{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 💸 Regression With An Insurance Dataset GBDT\nWelcome to the 2024 Kaggle Playground Series! We plan to continue in the spirit of previous playgrounds, providing interesting an approachable datasets for our community to practice their machine learning skills, and anticipate a competition each month.\n\n**Your Goal:** The objectives of this challenge is to predict insurance premiums based on various factors.","metadata":{"execution":{"iopub.status.busy":"2024-12-01T20:27:33.656191Z","iopub.execute_input":"2024-12-01T20:27:33.656758Z","iopub.status.idle":"2024-12-01T20:27:33.662726Z","shell.execute_reply.started":"2024-12-01T20:27:33.656719Z","shell.execute_reply":"2024-12-01T20:27:33.660821Z"}}},{"cell_type":"markdown","source":"# 1. Loading Libraries\nImport all the requiered libraries...","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Importing additional libraries\nfrom sklearn.preprocessing import LabelEncoder # label encoder function from sklearn\nfrom sklearn.model_selection import train_test_split, cross_val_score, KFold\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error, mean_squared_log_error","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Configuring the Notebook\nSet up some important parameters for the notebook...","metadata":{}},{"cell_type":"code","source":"%%time\n# I like to disable my Notebook Warnings.\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Configure notebook display settings to only use 2 decimal places, tables look nicer.\npd.options.display.float_format = '{:,.3f}'.format\npd.set_option('display.max_columns', 15) \npd.set_option('display.max_rows', 25)\n\n# Define some of the notebook parameters for future experiment replication.\nSEED   = 548","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Loading Datasets\nLoad the datasets into a pandas dataframe...","metadata":{}},{"cell_type":"code","source":"def load_csv_to_dataframe(file_path, ignore_fields=[]):\n    \"\"\"\n    Load a CSV file into a pandas DataFrame, optionally ignoring specified fields.\n\n    Parameters:\n    file_path (str): The file path of the CSV file to be loaded.\n    ignore_fields (list): A list of field names to be ignored when loading the CSV.\n\n    Returns:\n    pandas.DataFrame: A DataFrame containing the data from the CSV file, excluding the ignored fields.\n    \"\"\"\n    # Read the CSV file from the given file path using pandas\n    df = pd.read_csv(file_path)\n    \n    # Drop the fields that need to be ignored, if they exist in the DataFrame\n    df = df.drop(columns=ignore_fields, errors='ignore')\n    \n    # Return the resulting DataFrame\n    return df\n\n# Example usage:\n# df = load_csv_to_dataframe('data/sample.csv', ignore_fields=['column_to_ignore'])\n# print(df.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the competiion dataset\ntrn_input = '/kaggle/input/playground-series-s4e12/train.csv'\ntst_input = '/kaggle/input/playground-series-s4e12/test.csv'\nsub_input = '/kaggle/input/playground-series-s4e12/sample_submission.csv'\n\ntrn_df = load_csv_to_dataframe(trn_input, ignore_fields=['id'])\ntst_df = load_csv_to_dataframe(tst_input, ignore_fields=['id'])\nsub_df = load_csv_to_dataframe(sub_input)\n\n# Load the original dataset\n#org_input = '/kaggle/input/insurance-premium-prediction/Insurance Premium Prediction Dataset.csv'\n#org_df = load_csv_to_dataframe(org_input, ['id'])\n#org_df['is_original'] = 1\n#trn_df['is_original'] = 0\n#tst_df['is_original'] = 0\n#trn_df = pd.concat(objs=[trn_df, org_df])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Exploring The Dataset\nExploring the dataframe using a quick exploration function...","metadata":{}},{"cell_type":"code","source":"def extensive_eda(df):\n    \"\"\"\n    Perform exploratory data analysis (EDA) on the given DataFrame.\n\n    Parameters:\n    df (pandas.DataFrame): The DataFrame to analyze.\n\n    Returns:\n    None\n    \"\"\"\n    from IPython.display import display\n\n    # Display the DataFrame info\n    print(\"Information about the DataFrame:\")\n    df_info = df.info()\n    display(df_info)\n    print(\".....\")\n    print(\"\\n\")\n\n    # Display the first few rows of data\n    print(\"First few rows of the DataFrame:\")\n    display(df.head().T)\n    print(\".....\")\n    print(\"\\n\")\n    \n    # Display the number of duplicate values in each column\n    print(\"Number of duplicate values in each column:\")\n    duplicate_counts = df.duplicated().sum()\n    display(duplicate_counts)\n    print(\".....\")\n    print(\"\\n\")\n    \n    # Display the number of missing datapoints in each column\n    print(\"Number of missing datapoints in each column:\")\n    missing_counts = df.isna().sum()\n    display(missing_counts)\n    print(\".....\")\n    print(\"\\n\")\n    \n    # Display the number of outliers in each column using the IQR technique\n    print(\"Number of outliers in each column (using IQR technique):\")\n    outliers = {}\n    for column in df.select_dtypes(include=['number']).columns:\n        Q1 = df[column].quantile(0.25)\n        Q3 = df[column].quantile(0.75)\n        IQR = Q3 - Q1\n        outliers[column] = df[(df[column] < (Q1 - 1.5 * IQR)) | (df[column] > (Q3 + 1.5 * IQR))].shape[0]\n    display(outliers)\n    print(\".....\")\n    print(\"\\n\")\n    \n    # Display basic statistics of the DataFrame\n    print(\"Statistical summary of the DataFrame:\")\n    display(df.describe())\n    print(\".....\")\n    print(\"\\n\")\n    \n    # Display unique value count for each column\n    print(\"Number of unique values in each column:\")\n    unique_counts = df.nunique()\n    display(unique_counts)\n    print(\".....\")\n    print(\"\\n\")\n    \n    # Display column-wise summary in a table\n    print(\"Column-wise summary:\")\n    summary_data = []\n    for column in df.columns:\n        column_summary = {\n            \"Column\": column,\n            \"Data Type\": df[column].dtype,\n            \"Missing Values\": missing_counts[column],\n            \"Unique Values\": unique_counts[column],\n            \"Outliers\": outliers.get(column, 0) if df[column].dtype in ['int64', 'float64'] else \"N/A\",\n            \"Top 5 Values\": df[column].value_counts().head().to_dict() if df[column].dtype == 'object' else \"N/A\"\n        }\n        summary_data.append(column_summary)\n    summary_df = pd.DataFrame(summary_data)\n    display(summary_df.style.set_properties(**{'text-align': 'left'}).set_table_styles([dict(selector='th', props=[('text-align', 'left')])]))\n    print(\".....\")\n    print(\"\\n\")\n    \n    # Display correlation matrix for numerical features only\n    print(\"Correlation matrix of numerical features:\")\n    numerical_df = df.select_dtypes(include=['number'])\n    display(numerical_df.corr())\n    print(\".....\")\n    print(\"\\n\")\n    \n    # Display value counts for categorical columns in a table with widened format\n    print(\"Value counts for categorical columns:\")\n    value_counts_data = []\n    for column in df.select_dtypes(include=['object']).columns:\n        value_counts = df[column].value_counts().head().to_dict()\n        value_counts_data.append({\"Column\": column, \"Top 5 Values\": value_counts})\n    value_counts_df = pd.DataFrame(value_counts_data)\n    display(value_counts_df.style.set_properties(**{'text-align': 'left'}).set_table_styles([dict(selector='th', props=[('text-align', 'left')])]))\n    print(\".....\")\n    print(\"\\n\")\n\n# Example usage:\n# df = load_csv_to_dataframe('data/sample.csv')\n# perform_eda(df)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Quick EDA ...\nextensive_eda(trn_df)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Feature Engineering\nCreate features to improve model performance...","metadata":{}},{"cell_type":"code","source":"trn_df = trn_df.dropna(subset = ['Premium Amount'])\n#trn_df = trn_df.dropna()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the function to extract time value features\ndef extract_time_features(df, datetime_column):\n    # Ensure the column is in datetime format\n    df[datetime_column] = pd.to_datetime(df[datetime_column])\n    \n    # Extract various time features\n    df['year'] = df[datetime_column].dt.year\n    df['month'] = df[datetime_column].dt.month\n    df['day'] = df[datetime_column].dt.day\n    #df['hour'] = df[datetime_column].dt.hour\n    #df['minute'] = df[datetime_column].dt.minute\n    #df['second'] = df[datetime_column].dt.second\n    #df['microsecond'] = df[datetime_column].dt.microsecond\n    #df['day_of_week'] = df[datetime_column].dt.dayofweek  # Monday=0, Sunday=6\n    #df['day_name'] = df[datetime_column].dt.day_name()\n    #df['month_name'] = df[datetime_column].dt.month_name()\n    df['week_of_year'] = df[datetime_column].dt.isocalendar().week\n    #df['quarter'] = df[datetime_column].dt.quarter\n    #df['is_weekend'] = df[datetime_column].dt.dayofweek >= 5\n    #df['month_sin'] = np.sin(2 * np.pi * df['month'] / 12)\n    #df['month_cos'] = np.cos(2 * np.pi * df['month'] / 12)\n    #df['day_sin'] = np.sin(2 * np.pi * df['day'] / 31)\n    #df['day_cos'] = np.cos(2 * np.pi * df['day'] / 31)\n    \n    df = df.drop(columns = [datetime_column, \n                            #'day',\n                            #'month'\n                           ])\n\n    return df\n\n# Example usage\n# data = {'datetime': ['2023-12-23 15:21:39.134960', '2023-11-22 11:05:13.123456']}\n# df = pd.DataFrame(data)\n# df = extract_time_features(df, 'datetime')\n# print(df)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Creates time features\n#trn_df = extract_time_features(trn_df, 'Policy Start Date')\n#tst_df = extract_time_features(tst_df, 'Policy Start Date')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Fixing Missing Data\nUtilize imputing logic to fill missing values...","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\ndef impute_missing_values(train_df, test_df, target_column):\n    \"\"\"\n    Impute missing values for categorical and numerical columns in the training and test DataFrames.\n\n    Parameters:\n    train_df (pandas.DataFrame): The training DataFrame with missing values.\n    test_df (pandas.DataFrame): The testing DataFrame with missing values.\n    target_column (str): The name of the target column.\n\n    Returns:\n    tuple: A tuple containing the training and testing DataFrames with imputed values.\n    \"\"\"\n    # Create copies of the DataFrames to avoid modifying the originals\n    train_imputed = train_df.copy()\n    test_imputed = test_df.copy()\n    \n    # Separate categorical and numerical columns, excluding the target column\n    categorical_columns = train_imputed.select_dtypes(include=['object']).columns.difference([target_column])\n    numerical_columns = train_imputed.select_dtypes(include=['number']).columns.difference([target_column])\n    \n    # Impute missing values for categorical columns using the most frequent value\n    #cat_imputer = SimpleImputer(strategy='most_frequent')\n    cat_imputer = SimpleImputer(strategy='constant', fill_value='NAN')\n    train_imputed[categorical_columns] = cat_imputer.fit_transform(train_imputed[categorical_columns])\n    test_imputed[categorical_columns] = cat_imputer.transform(test_imputed[categorical_columns])\n    \n    # Impute missing values for numerical columns using the mean value\n    #num_imputer = SimpleImputer(strategy='mean')\n    num_imputer = SimpleImputer(strategy='constant', fill_value=-99)\n    train_imputed[numerical_columns] = num_imputer.fit_transform(train_imputed[numerical_columns])\n    test_imputed[numerical_columns] = num_imputer.transform(test_imputed[numerical_columns])\n    \n    return train_imputed, test_imputed\n\n# Example usage:\n# train_df = load_csv_to_dataframe('data/train.csv')\n# test_df = load_csv_to_dataframe('data/test.csv')\n# train_imputed, test_imputed = impute_missing_values(train_df, test_df, 'target')\n# print(train_imputed.head())\n# print(test_imputed.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Utilize the imputation function...\ntrain_imputed, test_imputed = impute_missing_values(trn_df, tst_df, \"Premium Amount\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_imputed.head().T","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a list of categorical variables to help the label encodin function\ncategorical_fields = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location', 'Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type', 'year']","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def label_encode_datasets(train_df, test_df, categ_fields):\n    \"\"\"\n    Label encode the categorical variables of the train and test DataFrames.\n\n    Parameters:\n    train_df (pandas.DataFrame): The training DataFrame.\n    test_df (pandas.DataFrame): The testing DataFrame.\n\n    Returns:\n    tuple: A tuple containing the label encoded training and testing DataFrames.\n    \"\"\"\n    # Create a copy of train and test dataframes to avoid modifying original dataframes\n    train_encoded = train_df.copy()\n    test_encoded = test_df.copy()\n    \n    # Identify categorical columns\n    # categorical_columns = test_encoded.select_dtypes(include=['object']).columns\n    categorical_columns = categ_fields\n    \n    # Initialize label encoder\n    le = LabelEncoder()\n    \n    # Apply label encoding to each categorical column\n    for column in categorical_columns:\n        print(f'Encoding: {column} ...')\n        # Fit the label encoder on the train data\n        le.fit(train_encoded[column])\n        \n        # Transform both train and test data using the same encoder\n        train_encoded[column] = le.transform(train_encoded[column])\n        if column in test_encoded.columns:\n            # Handle cases where test set may have unseen labels by using fillna\n            test_encoded[column] = test_encoded[column].map(lambda s: le.transform([s])[0] if s in le.classes_ else None)\n            test_encoded[column].fillna(-1, inplace=True)\n            test_encoded[column] = test_encoded[column].astype(int)\n\n    return train_encoded, test_encoded\n\n# Example usage:\n# train_df = load_csv_to_dataframe('data/train.csv')\n# test_df = load_csv_to_dataframe('data/test.csv')\n# train_encoded, test_encoded = label_encode_datasets(train_df, test_df)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encoding the train and test datasets.\ntrn_encoded, tst_encoded = label_encode_datasets(train_imputed, test_imputed, categorical_fields)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. ML Model Development","metadata":{}},{"cell_type":"code","source":"# Create a list of all the features available for training.\ntrn_encoded.columns","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to train an XGBoost Regressor with GPU support\ndef train_xgboost_regressor(train_df, test_df, target_column, param_file=None, n_splits=5):\n    \"\"\"\n    Train an XGBoost regressor using the provided training and test datasets with K-Fold cross-validation, utilizing GPU support.\n\n    Parameters:\n    train_df (pandas.DataFrame): The training DataFrame.\n    test_df (pandas.DataFrame): The testing DataFrame.\n    target_column (str): The name of the target column.\n    param_file (dict): Dictionary of hyperparameters for the XGBoost model.\n    n_splits (int): The number of folds for cross-validation.\n\n    Returns:\n    tuple: A tuple containing the model performance metrics and the predictions on the test dataset.\n    \"\"\"\n    from xgboost import XGBRegressor\n    from sklearn.model_selection import KFold\n    from sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score, mean_squared_log_error\n    import numpy as np\n\n    # Separate features and target from the training data\n    X = train_df.drop(columns=[target_column])\n    y = train_df[target_column]\n    \n    # Set default parameters if none are provided\n    if param_file is None:\n        param_file = {\n            'n_estimators': 2048,\n            'learning_rate': 0.02,\n            'max_depth': 6,\n            'subsample': 0.7,\n            'colsample_bytree': 0.7,\n            #'tree_method': 'gpu_hist',  # Enable GPU support\n            #'predictor': 'gpu_predictor',\n            'random_state': SEED\n        }\n\n    # Initialize the XGBoost model with parameters from param_file\n    model = XGBRegressor(**param_file)\n\n    # Initialize KFold cross-validation\n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\n\n    # Lists to store cross-validation results\n    mse_scores = []\n    mae_scores = []\n    r2_scores = []\n    rmsle_scores = []\n    test_predictions = []\n\n    # Perform K-Fold cross-validation\n\n    fold_number = 0\n    for train_index, val_index in kf.split(X):\n        fold_number += 1\n        #print('Fold:', fold_number)\n\n        X_train, X_val = X.iloc[train_index], X.iloc[val_index]\n        y_train, y_val = np.log1p(y.iloc[train_index]), np.log1p(y.iloc[val_index])\n\n        # Train the model\n        model.fit(X_train, y_train)\n\n        # Make predictions on the validation set\n        y_val_pred = model.predict(X_val)\n\n        # Calculate validation metrics\n        mse_scores.append(mean_squared_error(y_val, y_val_pred))\n        mae_scores.append(mean_absolute_error(y_val, y_val_pred))\n        r2_scores.append(r2_score(y_val, y_val_pred))\n        #rmsle_scores.append(np.sqrt(mean_squared_log_error(y_val, np.maximum(y_val_pred, 0))))\n\n        # Make predictions on the test dataset for each fold\n        X_test = test_df.drop(columns=[target_column], errors='ignore')\n        test_predictions.append(model.predict(X_test))\n\n        print(f'Fold {fold_number} RMSLE = {mse_scores[fold_number-1]}')\n\n    # Calculate average metrics\n    avg_mse = np.mean(mse_scores)\n    avg_mae = np.mean(mae_scores)\n    avg_r2 = np.mean(r2_scores)\n    #avg_rmsle = np.mean(rmsle_scores)\n\n    # Print the metrics in a readable format\n    print(\"Model Performance Metrics (Cross-Validation):\")\n    print(\"..................\")\n    print(f\"Average MSE: {avg_mse:.4f}\")\n    print(f\"Average MAE: {avg_mae:.4f}\")\n    print(f\"Average R2 Score: {avg_r2:.4f}\")\n    #print(f\"Average RMSLE: {avg_rmsle:.4f}\")\n\n    # Calculate the average predictions across all folds\n    y_test_pred = np.mean(test_predictions, axis=0)\n\n    return y_test_pred\n\n# Example usage:\n# param_file = {\n#     'n_estimators': 200,\n#     'learning_rate': 0.05,\n#     'max_depth': 8,\n#     'tree_method': 'gpu_hist',  # Enable GPU support\n#     'predictor': 'gpu_predictor'\n# }\n# train_df = load_csv_to_dataframe('data/train.csv')\n# test_df = load_csv_to_dataframe('data/test.csv')\n# test_predictions = train_xgboost_regressor(train_df, test_df, 'target', param_file=param_file)\n# print(test_predictions)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions = train_xgboost_regressor(trn_encoded, tst_encoded, 'Premium Amount')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Average RMSLE: 1.1738 Plain Model Using Desicion Trees\n# Average RMSLE: 1.1948\n# Average MSE: 1.1033","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the predictions to the Submission Dataset...\nsub_df['Premium Amount'] = np.expm1(test_predictions)\n\n# Save submission to CSV\nsub_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sub_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to train and blend predictions from XGBoost, LightGBM, CatBoost, Linear Regression, SVR, and Neural Network\n\ndef train_blend_models(train_df, test_df, target_column, param_files=None, n_splits=5, weights=None):\n    \"\"\"\n    Train XGBoost, LightGBM, CatBoost, Linear Regression, SVR, and TensorFlow-based Neural Network regressors using the provided training and test datasets with K-Fold cross-validation.\n    Optimize the weights for blending based on out-of-fold validation predictions to maximize performance.\n\n    Parameters:\n    train_df (pandas.DataFrame): The training DataFrame.\n    test_df (pandas.DataFrame): The testing DataFrame.\n    target_column (str): The name of the target column.\n    param_files (dict): Dictionary containing hyperparameters for each model.\n    n_splits (int): The number of folds for cross-validation.\n    weights (list): Initial weights for blending the predictions from each model (optional).\n\n    Returns:\n    np.ndarray: Weighted blended predictions on the test dataset.\n    \"\"\"\n    from xgboost import XGBRegressor\n    from lightgbm import LGBMRegressor\n    from catboost import CatBoostRegressor\n    from sklearn.linear_model import LinearRegression\n    from sklearn.svm import SVR\n    from tensorflow.keras.models import Sequential\n    from tensorflow.keras.layers import Dense\n    from tensorflow.keras.optimizers import Adam\n    from sklearn.model_selection import KFold\n    from sklearn.metrics import mean_squared_error\n    from scipy.optimize import minimize\n    import numpy as np\n    import pandas as pd\n\n    # Separate features and target from the training data\n    X = train_df.drop(columns=[target_column])\n    y = train_df[target_column]\n\n    # Set default hyperparameters if none are provided\n    if param_files is None:\n        param_files = {\n            'xgboost': {\n                'n_estimators': 256,\n                'learning_rate': 0.05,\n                'max_depth': 12,\n                'subsample': 0.8,\n                'colsample_bytree': 0.8,\n                'random_state': 42,\n                'tree_method': 'gpu_hist',\n                'predictor': 'gpu_predictor'\n            },\n            'lightgbm': {\n                'n_estimators': 256,\n                'learning_rate': 0.05,\n                'max_depth': 12,\n                'num_leaves': 31,\n                'random_state': 42,\n                'verbose': -1\n            },\n            'catboost': {\n                'iterations': 256,\n                'learning_rate': 0.05,\n                'depth': 12,\n                'random_seed': 42,\n                'silent': True,\n                'task_type': 'GPU'\n            },\n            'svr': {\n                'kernel': 'linear',\n                'C': 1.0,\n                'epsilon': 0.1\n            },\n            'mlp': {\n                'hidden_layer_sizes': (64, 32),\n                'activation': 'relu',\n                'solver': 'adam',\n                'max_iter': 500,\n                'random_state': 42\n            }\n        }\n\n    # TensorFlow-based Neural Network Model\n    def create_nn_model(input_dim):\n        model = Sequential([\n            Dense(64, activation='relu', input_dim=input_dim),\n            Dense(32, activation='relu'),\n            Dense(1)  # Single output for regression\n        ])\n        model.compile(optimizer=Adam(learning_rate=0.001), loss='mse')\n        return model\n\n    # Initialize models\n    models = {\n        'xgboost': XGBRegressor(**param_files['xgboost']),\n        'lightgbm': LGBMRegressor(**param_files['lightgbm']),\n        'catboost': CatBoostRegressor(**param_files['catboost']),\n        #'linear_regression': LinearRegression(),\n        #'svr': SVR(**param_files['svr']),\n        #'neural_network': create_nn_model(X.shape[1])\n    }\n\n    # Initialize KFold cross-validation\n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\n\n    # Store out-of-fold predictions and validation actuals\n    oof_predictions = {model_name: np.zeros(len(train_df)) for model_name in models.keys()}\n    blended_val_actuals = np.zeros(len(train_df))\n    test_predictions = {model_name: [] for model_name in models.keys()}\n    mse_scores = {model_name: [] for model_name in models.keys()}\n\n    # Perform K-Fold cross-validation for each model\n    for train_index, val_index in kf.split(X):\n        X_train, X_val = X.iloc[train_index], X.iloc[val_index]\n        y_train, y_val = np.log1p(y.iloc[train_index]), np.log1p(y.iloc[val_index])\n\n        for model_name, model in models.items():\n            print(f'Training Model {model_name} ...')\n            if model_name == 'neural_network':\n                # Train TensorFlow model\n                model.fit(X_train, y_train, epochs=50, batch_size=32, verbose=0, validation_data=(X_val, y_val))\n                # Predict on validation set\n                oof_predictions[model_name][val_index] = model.predict(X_val).flatten()\n                # Predict on test dataset\n                X_test = test_df.drop(columns=[target_column], errors='ignore')\n                test_predictions[model_name].append(model.predict(X_test).flatten())\n            else:\n                # Train other models\n                model.fit(X_train, y_train)\n                # Predict on validation set\n                oof_predictions[model_name][val_index] = model.predict(X_val)\n                # Predict on test dataset\n                X_test = test_df.drop(columns=[target_column], errors='ignore')\n                test_predictions[model_name].append(model.predict(X_test))\n\n        blended_val_actuals[val_index] = y_val\n\n    # Print average MSE for each model\n    print(\"Model Performance on Validation Data:\")\n    for model_name, scores in mse_scores.items():\n        avg_mse = np.mean(scores)\n        print(f\"{model_name}: Average MSE = {avg_mse:.4f}\")\n\n    # Define the optimization function to minimize blended validation MSE\n    def optimize_weights(weights):\n        blended_preds = np.zeros(len(train_df))\n        for i, model_name in enumerate(models.keys()):\n            blended_preds += weights[i] * oof_predictions[model_name]\n        return mean_squared_error(blended_val_actuals, blended_preds)\n\n    # Set initial weights and bounds for optimization\n    initial_weights = np.ones(len(models)) / len(models)\n    bounds = [(0, 1) for _ in models.keys()]\n    constraints = ({'type': 'eq', 'fun': lambda w: 1 - sum(w)})\n\n    # Perform optimization\n    optimized_result = minimize(optimize_weights, initial_weights, bounds=bounds, constraints=constraints)\n    optimal_weights = optimized_result.x\n    print(f\"Optimized Weights: {optimal_weights}\")\n\n    # Blend out-of-fold predictions using the optimized weights\n    blended_val_predictions = np.zeros(len(train_df))\n    for i, model_name in enumerate(models.keys()):\n        blended_val_predictions += optimal_weights[i] * oof_predictions[model_name]\n\n    blended_val_mse = mean_squared_error(blended_val_actuals, blended_val_predictions)\n    print(f\"Blended Model Validation MSE: {blended_val_mse:.4f}\")\n\n    # Calculate the average predictions across all folds for each model\n    avg_test_predictions = {\n        model_name: np.mean(preds, axis=0)\n        for model_name, preds in test_predictions.items()\n    }\n\n    # Blend the predictions using the optimized weights\n    blended_predictions = np.zeros(len(test_df))\n    for i, model_name in enumerate(models.keys()):\n        blended_predictions += optimal_weights[i] * avg_test_predictions[model_name]\n\n    return blended_predictions\n\n# Example usage:\n# param_files = {\n#     'xgboost': {\n#         'n_estimators': 256,\n#         'learning_rate': 0.05,\n#         'max_depth': 12,\n#         'tree_method': 'gpu_hist',\n#         'predictor': 'gpu_predictor'\n#     },\n#     'lightgbm': {\n#         'n_estimators': 256,\n#         'learning_rate': 0.05,\n#         'max_depth': 12,\n#         'num_leaves': 64\n#     },\n#     'catboost': {\n#         'iterations': 256,\n#         'learning_rate': 0.05,\n#         'depth': 12\n#     },\n#     'svr': {\n#         'kernel': 'rbf',\n#         'C': 1.0,\n#         'epsilon': 0.1\n#     },\n#     'mlp': {\n#         'hidden_layer_sizes': (64, 32),\n#         'activation': 'relu',\n#         'solver': 'adam',\n#         'max_iter': 500\n#     }\n# }\n# train_df = load_csv_to_dataframe('data/train.csv')\n# test_df = load_csv_to_dataframe('data/test.csv')\n# blended_predictions = train_blend_models(train_df, test_df, 'target', param_files=param_files)\n# print(blended_predictions)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#blended_predictions = train_blend_models(trn_encoded, tst_encoded, 'Premium Amount')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Load the predictions to the Submission Dataset...\n# sub_df['Premium Amount'] = np.expm1(blended_predictions)\n\n# #Save submission to CSV\n# sub_df.to_csv('blended_submission.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}