{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import library","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.preprocessing import StandardScaler\nfrom collections import Counter\nfrom sklearn.feature_selection import mutual_info_regression, SelectKBest, f_regression\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_log_error\nimport logging\nimport lightgbm as lgb\nimport xgboost as xgb\nimport catboost as cb\n\nimport gc\n#Ignore warnings\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:25.026458Z","iopub.execute_input":"2024-12-21T02:30:25.026828Z","iopub.status.idle":"2024-12-21T02:30:25.031614Z","shell.execute_reply.started":"2024-12-21T02:30:25.026799Z","shell.execute_reply":"2024-12-21T02:30:25.030655Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data load","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\n\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:25.033227Z","iopub.execute_input":"2024-12-21T02:30:25.033531Z","iopub.status.idle":"2024-12-21T02:30:30.942373Z","shell.execute_reply.started":"2024-12-21T02:30:25.033486Z","shell.execute_reply":"2024-12-21T02:30:30.941485Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n<span style=\"font-size:20px;\">Set id as index</span>","metadata":{}},{"cell_type":"code","source":"train.set_index('id', inplace=True)\ntest.set_index('id', inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:31.47973Z","iopub.execute_input":"2024-12-21T02:30:31.480036Z","iopub.status.idle":"2024-12-21T02:30:31.485453Z","shell.execute_reply.started":"2024-12-21T02:30:31.480009Z","shell.execute_reply":"2024-12-21T02:30:31.484516Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature engineering","metadata":{}},{"cell_type":"markdown","source":"## Data classification","metadata":{}},{"cell_type":"markdown","source":"\n<span style=\"font-size:18px;\">Categorize the data for further processing</span>","metadata":{}},{"cell_type":"code","source":"\nnum_cols = list(train.select_dtypes(include=['number']).columns)\ncat_cols = list(train.select_dtypes(exclude=['number', 'datetime64[ns]']).columns)\ndatetime_cols = ['Policy Start Date']\n\nif 'Premium Amount' in num_cols:\n    num_cols.remove('Premium Amount')\nif 'Policy Start Date' in cat_cols:\n    cat_cols.remove('Policy Start Date')\n\nprint(\"Numerical columns:\")\nprint(num_cols)\nprint(\"\\nCategorical columns excluding datetime columns:\")\nprint(cat_cols)\nprint(\"\\nDatetime column:\")\nprint(datetime_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:31.486429Z","iopub.execute_input":"2024-12-21T02:30:31.486719Z","iopub.status.idle":"2024-12-21T02:30:32.084239Z","shell.execute_reply.started":"2024-12-21T02:30:31.486691Z","shell.execute_reply":"2024-12-21T02:30:32.083174Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handling of missing values","metadata":{}},{"cell_type":"markdown","source":"<span style=\"font-size:18px;\">In the processing of missing values, I tried to fill in the mean and median values. But the effect is not as good as this treatment</span>\n\n\n<span style=\"font-size:18px;\">**So I'm just going to show you this treatment here**</span>","metadata":{}},{"cell_type":"code","source":"\ndef fill_missing_values(df, num_cols, cat_cols):\n    \n    # The missing value for filling numerical features is -1\n    for col in num_cols:\n        if col in df.columns:\n            df[col] = df[col].fillna(-1)\n            \n    # Missing value of fill classification feature is 'Unknown'\n    for col in cat_cols:\n        if col in df.columns:\n            df[col] = df[col].fillna('Unknown')\n    \n    return df\n\ntrain = fill_missing_values(train, num_cols, cat_cols)\ntest = fill_missing_values(test, num_cols, cat_cols)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:32.085575Z","iopub.execute_input":"2024-12-21T02:30:32.085861Z","iopub.status.idle":"2024-12-21T02:30:33.230736Z","shell.execute_reply.started":"2024-12-21T02:30:32.085833Z","shell.execute_reply":"2024-12-21T02:30:33.229716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:33.231868Z","iopub.execute_input":"2024-12-21T02:30:33.232145Z","iopub.status.idle":"2024-12-21T02:30:33.772565Z","shell.execute_reply.started":"2024-12-21T02:30:33.232119Z","shell.execute_reply":"2024-12-21T02:30:33.771617Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Datetime feature","metadata":{}},{"cell_type":"markdown","source":"<span style=\"font-size:20px;\">In the date-time processing, I tried all of the following comments</span>\n\n\n\nI found that if I processed the month and day, it would affect my performance\n\n**The results showed that only the year treatment, the best effect**\n\nGive it a try if you're interested\n\n<span style=\"font-size:15px;\">**CV score 1.0460 , LB score 1.0453**</span>","metadata":{}},{"cell_type":"code","source":"\ndef process_date_features(df, date_col):\n    df[date_col] = pd.to_datetime(df[date_col])\n    \n    df['Year'] = df[date_col].dt.year\n    # df['Month'] = df[date_col].dt.month\n    # df['Day'] = df[date_col].dt.day\n    \n    df['YearSin'] = np.sin(2 * np.pi * df['Year'] / 4)\n    df['YearCos'] = np.cos(2 * np.pi * df['Year'] / 4)\n    # df['MonthSin'] = np.sin(2 * np.pi * df['Month'] / 12)\n    # df['MonthCos'] = np.cos(2 * np.pi * df['Month'] / 12)\n    # df['DaySin'] = np.sin(2 * np.pi * df['Day'] / 30)\n    # df['DayCos'] = np.cos(2 * np.pi * df['Day'] / 30)\n    \n    # df['Season'] = df['Month'].apply(lambda x: 'Winter' if x in [12, 1, 2] else\n    #                                            'Spring' if x in [3, 4, 5] else\n    #                                            'Summer' if x in [6, 7, 8] else\n    #                                            'Autumn')\n    # df = pd.get_dummies(df, columns=['Season'], prefix=['Season'])\n    df = df.drop(date_col, axis=1)\n    \n    return df\n\ntrain = process_date_features(train, 'Policy Start Date')\ntest = process_date_features(test, 'Policy Start Date')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<span style=\"font-size:15px;\">I also tried to deal with it in more detail here, but it didn't improve the score</span>\n\n\n<span style=\"font-size:15px;\">**CV score 1.0461**</span>","metadata":{}},{"cell_type":"code","source":"\"\"\"\n\ndef process_date_features(df):\n    \n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Month'] = df['Policy Start Date'].dt.month\n    df['Monthname'] = df['Policy Start Date'].dt.month_name()\n    df['Dayofweek'] = df['Policy Start Date'].dt.day_name()\n    df['Week'] = df['Policy Start Date'].dt.isocalendar().week\n    df['Yearsin'] = np.sin(2 * np.pi * df['Year'])\n    df['Yearcos'] = np.cos(2 * np.pi * df['Year'])\n    df['Monthsin'] = np.sin(2 * np.pi * df['Month'] / 12) \n    df['Monthcos'] = np.cos(2 * np.pi * df['Month'] / 12)\n    df['Daysin'] = np.sin(2 * np.pi * df['Day'] / 31)  \n    df['Daycos'] = np.cos(2 * np.pi * df['Day'] / 31)\n    \n    df.drop('Policy Start Date', axis=1, inplace=True)\n\n    return df\n\ntrain = process_date_features(train)\ntest = process_date_features(test)\n\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:33.773655Z","iopub.execute_input":"2024-12-21T02:30:33.773931Z","iopub.status.idle":"2024-12-21T02:30:33.780178Z","shell.execute_reply.started":"2024-12-21T02:30:33.773904Z","shell.execute_reply":"2024-12-21T02:30:33.779221Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Numerical feature processing","metadata":{}},{"cell_type":"markdown","source":"<span style=\"font-size:18px;\">I tried to standardize the numerical features, but the scores dropped</span>\n","metadata":{}},{"cell_type":"code","source":"\"\"\"\n\ndef standardize_numeric_features(df, num_cols=None):\n    scaler = StandardScaler()\n    for col in num_cols:\n        if col in df.columns:\n            df[col] = scaler.fit_transform(df[col].values.reshape(-1, 1))\n    return df\n\nnum_cols = ['Age', 'Annual Income', 'Health Score', 'Credit Score']\ntrain = standardize_numeric_features(train, num_cols)\ntest = standardize_numeric_features(test, num_cols)\n\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:34.917808Z","iopub.execute_input":"2024-12-21T02:30:34.918082Z","iopub.status.idle":"2024-12-21T02:30:34.923885Z","shell.execute_reply.started":"2024-12-21T02:30:34.918058Z","shell.execute_reply":"2024-12-21T02:30:34.92285Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Classification feature transformation","metadata":{}},{"cell_type":"markdown","source":"<span style=\"font-size:20px;\">For the processing of classification features, I used tag coding, one-hot coding, binary coding, ordinal coding, target coding, frequency coding and manual coding for comparison</span>\n\n<span style=\"font-size:15px;\">Here I only show frequency coding versus manual coding. Because through my score comparison, I found that these two methods are better.</span>\n\n\n<span style=\"font-size:20px;\">**Relatively speaking, frequency coding performs best.**</span>\n\nYou can also use manual coding if you want to experiment","metadata":{}},{"cell_type":"code","source":"# Remove the comment if you use the second date-time processing method\n#cat_cols.extend(['Monthname', 'Dayofweek'])\ndef frequency_encode(df, columns):\n    encoded_df = df.copy()\n    if 'Policy Start Date' in df.columns:\n        df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n        # Add the Monthname and Dayofweek columns\n        #df['Monthname'] = df['Policy Start Date'].dt.month_name()\n        #df['Dayofweek'] = df['Policy Start Date'].dt.day_name()\n\n    for col in columns:\n        frequency = df[col].value_counts(normalize=True).to_dict()\n        new_col_name = f\"{col}_Freq_Enc\"\n        encoded_df[new_col_name] = df[col].map(frequency)\n\n    encoded_df = encoded_df.drop(columns=columns)\n    return encoded_df\n\ntrain = frequency_encode(train, cat_cols)\ntest = frequency_encode(test, cat_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:34.924884Z","iopub.execute_input":"2024-12-21T02:30:34.92514Z","iopub.status.idle":"2024-12-21T02:30:38.214701Z","shell.execute_reply.started":"2024-12-21T02:30:34.925115Z","shell.execute_reply":"2024-12-21T02:30:38.213646Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<span style=\"font-size:18px;\">In manual coding, I coded each classification feature, which I thought would help the model recognize better</span>\n\n\n**However, in the LB score, the performance is not as good as frequency coding**","metadata":{}},{"cell_type":"code","source":" \"\"\"\n \ndef encode_categorical_features(df):\n    \n    education_map = {\"High School\": 0, \"Bachelor's\": 1, \"Master's\": 2, \"PhD\": 3}\n    policy_map = {'Basic': 0, 'Comprehensive': 1, 'Premium': 2}\n    exercise_map = {'Rarely': 0, 'Daily': 1, 'Weekly': 2, 'Monthly': 3}\n    feedback_map = {'Poor': 0, 'Average': 1, 'Good': 2, \"Unknown\": 3}\n    gender_map = {'Male': 0, 'Female': 1}\n    smoking_map = {'Yes': 1, 'No': 0}\n    Occupation_map = {'Unemployed': 0, 'Self-Employed': 1, 'Employed': 2, 'Unknown': 3}\n    Marital_map = {'Divorced': 0, 'Married': 1, 'Single': 2, \"Unknown\": 3}  \n    Location_map = {'Urban': 0, 'Rural': 1, 'Suburban': 2}\n    Property_map = {'Basic': 0, 'Comprehensive': 1, 'Premium': 2}\n\n\n    df['Education Level'] = df['Education Level'].map(education_map)\n    df['Policy Type'] = df['Policy Type'].map(policy_map)\n    df['Exercise Frequency'] = df['Exercise Frequency'].map(exercise_map)\n    df['Customer Feedback'] = df['Customer Feedback'].map(feedback_map)\n    df['Gender'] = df['Gender'].map(gender_map)\n    df['Smoking Status'] = df['Smoking Status'].map(smoking_map)\n    df['Occupation'] = df['Occupation'].map(Occupation_map)\n    df['Marital Status'] = df['Marital Status'].map(Marital_map)\n    df['Location'] = df['Location'].map(Location_map)\n    df['Property Type'] = df['Property Type'].map(Property_map)\n    \n    return df\n\ntrain = encode_categorical_features(train)\ntest = encode_categorical_features(test)\n\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:38.215805Z","iopub.execute_input":"2024-12-21T02:30:38.21608Z","iopub.status.idle":"2024-12-21T02:30:38.222932Z","shell.execute_reply.started":"2024-12-21T02:30:38.216054Z","shell.execute_reply":"2024-12-21T02:30:38.222102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:38.224052Z","iopub.execute_input":"2024-12-21T02:30:38.224322Z","iopub.status.idle":"2024-12-21T02:30:38.258805Z","shell.execute_reply.started":"2024-12-21T02:30:38.224296Z","shell.execute_reply":"2024-12-21T02:30:38.257882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:38.259999Z","iopub.execute_input":"2024-12-21T02:30:38.260665Z","iopub.status.idle":"2024-12-21T02:30:38.28466Z","shell.execute_reply.started":"2024-12-21T02:30:38.260622Z","shell.execute_reply":"2024-12-21T02:30:38.283687Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature interaction","metadata":{}},{"cell_type":"markdown","source":"\n<span style=\"font-size:20px;\">Here I use health scores to interact with numerical features</span>\n\n**If you do this step, your CV score is 1.0461**\n\n**Without this step, the CV score is 1.0460**","metadata":{}},{"cell_type":"code","source":"\"\"\"\n\ndef feature_interaction(df, num_cols, health_score_column='Health Score'):\n    if health_score_column not in df.columns:\n        raise ValueError(f\"'{health_score_column}' column does not exist in the dataframe.\")\n\n    print(\"Available columns:\", df.columns.tolist())\n\n    num_cols = [col for col in num_cols if col in df.columns]\n\n    interaction_df = df.copy()\n\n    for col in num_cols:\n        if col!= health_score_column:\n            interaction_name = f\"{health_score_column}_x_{col}\"\n            interaction_df[interaction_name] = df[health_score_column] * df[col]\n\n            difference_name = f\"{health_score_column}_minus_{col}\"\n            interaction_df[difference_name] = df[health_score_column] - df[col]\n\n            ratio_name = f\"{health_score_column}_div_{col}\"\n            interaction_df[ratio_name] = df[health_score_column] / (df[col] + 1e-8)  \n\n    return interaction_df\n\n\ntrain_interacted = feature_interaction(train, num_cols=num_cols, health_score_column='Health Score')\ntest_interacted = feature_interaction(test, num_cols=num_cols, health_score_column='Health Score')\n\nprint(\"Train dataset shape after interaction:\", train_interacted.shape)\nprint(\"Test dataset shape after interaction:\", test_interacted.shape)\n\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:33:04.384883Z","iopub.execute_input":"2024-12-21T02:33:04.385757Z","iopub.status.idle":"2024-12-21T02:33:04.718797Z","shell.execute_reply.started":"2024-12-21T02:33:04.385721Z","shell.execute_reply":"2024-12-21T02:33:04.7179Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature selection","metadata":{}},{"cell_type":"markdown","source":"\n<span style=\"font-size:20px;\">Use mutual information for feature selection</span>\n\n**Here I chose the top 20 features based on their correlation to the target variable. But finding out by scoring doesn't help much**\n\n**The score of LB with feature selection was consistent with that of LB without feature selection**\n\n*You can also try it*","metadata":{}},{"cell_type":"code","source":"\"\"\"\n\ndef select_highly_correlated_features(train, test, target_column, k=20):\n    if target_column not in train.columns:\n        raise ValueError(f\"object column '{target_column}' Does not exist in the training set data box\")\n\n    X_train = train.drop(columns=[target_column])\n    y_train = train[target_column]\n\n    mi = mutual_info_regression(X_train, y_train)\n    mi_df = pd.DataFrame({'feature': X_train.columns, 'mi_score': mi})\n    mi_df = mi_df.sort_values(by='mi_score', ascending=False)\n\n    selected_features_mi = mi_df['feature'].tolist()[:k]\n\n    selector = SelectKBest(score_func=f_regression, k=k)\n    X_new = selector.fit_transform(X_train, y_train)\n    selected_features_f = X_train.columns[selector.get_support()].tolist()\n\n    selected_features = list(set(selected_features_mi + selected_features_f))\n\n    if len(selected_features) < k:\n        remaining_features = [col for col in X_train.columns if col not in selected_features]\n        selected_features.extend(remaining_features[:k - len(selected_features)])\n\n    test = test[selected_features]\n    X_train_selected = X_train[selected_features]\n    train_selected = pd.concat([X_train_selected, y_train], axis=1)\n\n    return train_selected, test\n\n\ntrain_selected, test_selected = select_highly_correlated_features(train, test, 'Premium Amount', k=20)\n\nprint(\"Selected Features:\")\nprint(train_selected.columns.tolist())\n\n\ntrain = train_selected\ntest = test_selected\n\nprint(\"Train dataset shape:\", train.shape)\nprint(\"Test dataset shape:\", test.shape)\n\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:39.627947Z","iopub.status.idle":"2024-12-21T02:30:39.628269Z","shell.execute_reply.started":"2024-12-21T02:30:39.628121Z","shell.execute_reply":"2024-12-21T02:30:39.628136Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model comparison","metadata":{}},{"cell_type":"markdown","source":"## LGB","metadata":{}},{"cell_type":"code","source":"\nlogging.getLogger('lightgbm').setLevel(logging.ERROR)  \nwarnings.filterwarnings(\"ignore\", category=UserWarning, message=\".*Found whitespace in feature_names.*\")\n\n\ny = train['Premium Amount']\nX_train = train.drop('Premium Amount', axis=1)\ny_log = np.log1p(y)\ntest = test[X_train.columns]\n\nparams = {\n    \"objective\": \"regression\",\n    \"metric\": \"rmse\",\n    \"seed\": 42,\n    \"verbose\": -1,  \n}\n\nn_splits = 5\nkf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\ncv_scores = []\noof_predictions = np.zeros(len(X_train))\nlgb_test_predictions = np.zeros(len(test))  \nbest_model = None\n\nfor train_index, val_index in kf.split(X_train):\n    X_train_fold, X_val_fold = X_train.iloc[train_index], X_train.iloc[val_index]\n    y_train_fold, y_val_fold = y_log.iloc[train_index], y_log.iloc[val_index]  \n\n    dtrain = lgb.Dataset(X_train_fold, label=y_train_fold)\n    dval = lgb.Dataset(X_val_fold, label=y_val_fold, reference=dtrain)\n\n    early_stopping = lgb.early_stopping(stopping_rounds=50, verbose=False)\n\n    model = lgb.train(\n        params,\n        dtrain,\n        valid_sets=[dval],\n        callbacks=[early_stopping],  \n    )\n\n    best_model = model\n\n    oof_predictions[val_index] = np.expm1(model.predict(X_val_fold))\n\n    lgb_test_predictions += np.expm1(model.predict(test)) / n_splits  \n\n    y_val_pred = model.predict(X_val_fold)\n    fold_rmsle = np.sqrt(mean_squared_log_error(y_val_fold, np.maximum(y_val_pred, 0)))\n    cv_scores.append(fold_rmsle)\n\nprint(f\"Mean RMSLE across {n_splits} folds: {np.mean(cv_scores):.4f}\")\nlgb_rmsle = np.sqrt(mean_squared_log_error(y, np.maximum(oof_predictions, 0)))\nprint(f\"Final RMSLE on out-of-fold predictions: {lgb_rmsle:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:33:33.653184Z","iopub.execute_input":"2024-12-21T02:33:33.654051Z","iopub.status.idle":"2024-12-21T02:34:08.108765Z","shell.execute_reply.started":"2024-12-21T02:33:33.654015Z","shell.execute_reply":"2024-12-21T02:34:08.107279Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## XGB","metadata":{}},{"cell_type":"code","source":"\nlogging.getLogger('xgboost').setLevel(logging.ERROR)  \nwarnings.filterwarnings(\"ignore\", category=UserWarning, message=\".*Found whitespace in feature_names.*\")\n\nparams = {\n    \"objective\": \"reg:squarederror\",\n    \"eval_metric\": \"rmse\",\n    \"seed\": 42,\n    \"verbosity\": 0,  \n}\n\nn_splits = 5\nkf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\ncv_scores = []\noof_predictions = np.zeros(len(X_train))\nxgb_test_predictions = np.zeros(len(test))  \nbest_model = None\n\nfor train_index, val_index in kf.split(X_train):\n    X_train_fold, X_val_fold = X_train.iloc[train_index], X_train.iloc[val_index]\n    y_train_fold, y_val_fold = y_log.iloc[train_index], y_log.iloc[val_index]  \n\n    dtrain = xgb.DMatrix(X_train_fold, label=y_train_fold)\n    dval = xgb.DMatrix(X_val_fold, label=y_val_fold)\n\n    early_stopping = xgb.callback.EarlyStopping(rounds=50, metric_name='rmse', data_name='validation')\n\n    model = xgb.train(\n        params,\n        dtrain,\n        num_boost_round=1000,\n        evals=[(dval, 'validation')],\n        callbacks=[early_stopping],  \n        verbose_eval=False,  \n    )\n\n    best_model = model\n\n    oof_predictions[val_index] = np.expm1(model.predict(dval))\n\n    xgb_test_predictions += np.expm1(model.predict(xgb.DMatrix(test))) / n_splits  \n\n    y_val_pred = model.predict(dval)\n    fold_rmsle = np.sqrt(mean_squared_log_error(y_val_fold, np.maximum(y_val_pred, 0)))\n    cv_scores.append(fold_rmsle)\n\nprint(f\"Mean RMSLE across {n_splits} folds: {np.mean(cv_scores):.4f}\")\nxgb_rmsle = np.sqrt(mean_squared_log_error(y, np.maximum(oof_predictions, 0)))\nprint(f\"Final RMSLE on out-of-fold predictions: {xgb_rmsle:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:39.631858Z","iopub.status.idle":"2024-12-21T02:30:39.632312Z","shell.execute_reply.started":"2024-12-21T02:30:39.632076Z","shell.execute_reply":"2024-12-21T02:30:39.632098Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CAT","metadata":{}},{"cell_type":"code","source":"\nlogging.getLogger('catboost').setLevel(logging.ERROR)  \nwarnings.filterwarnings(\"ignore\", category=UserWarning, message=\".*Found whitespace in feature_names.*\")\n\n\ny = train['Premium Amount']\nX_train = train.drop('Premium Amount', axis=1)\ny_log = np.log1p(y)\ntest = test[X_train.columns]\n\n\nparams = {\n    \"loss_function\": \"RMSE\",\n    \"random_seed\": 42,\n    \"verbose\": False,\n}\n\nn_splits = 5\nkf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\ncv_scores = []\noof_predictions = np.zeros(len(X_train))\ncat_test_predictions = np.zeros(len(test))  \nbest_model = None\n\nfor train_index, val_index in kf.split(X_train):\n    X_train_fold, X_val_fold = X_train.iloc[train_index], X_train.iloc[val_index]\n    y_train_fold, y_val_fold = y_log.iloc[train_index], y_log.iloc[val_index]  \n\n    dtrain = cb.Pool(X_train_fold, label=y_train_fold)\n    dval = cb.Pool(X_val_fold, label=y_val_fold)\n\n    model = cb.CatBoostRegressor(**params)\n    model.fit(\n        dtrain,\n        eval_set=dval,\n        early_stopping_rounds=50,\n        verbose=False,\n    )\n\n    best_model = model\n\n    oof_predictions[val_index] = np.expm1(model.predict(X_val_fold))\n\n    cat_test_predictions += np.expm1(model.predict(test)) / n_splits  \n\n    y_val_pred = model.predict(X_val_fold)\n    fold_rmsle = np.sqrt(mean_squared_log_error(y_val_fold, np.maximum(y_val_pred, 0)))\n    cv_scores.append(fold_rmsle)\n\nprint(f\"Mean RMSLE across {n_splits} folds: {np.mean(cv_scores):.4f}\")\ncat_rmsle = np.sqrt(mean_squared_log_error(y, np.maximum(oof_predictions, 0)))\nprint(f\"Final RMSLE on out-of-fold predictions: {cat_rmsle:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:34:40.850229Z","iopub.execute_input":"2024-12-21T02:34:40.850626Z","iopub.status.idle":"2024-12-21T02:37:16.92395Z","shell.execute_reply.started":"2024-12-21T02:34:40.850594Z","shell.execute_reply":"2024-12-21T02:37:16.923018Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"markdown","source":"\n<span style=\"font-size:24px;\">You can use this code to submit the model separately for verification</span>","metadata":{}},{"cell_type":"code","source":"\"\"\"\ntest_submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv') \n# finally_test_predictions = lgb_test_predictions or xgb_test_predictions or cat_test_predictions\nsubmission = pd.DataFrame({'id': test_submission['id'], 'Premium Amount': finally_test_predictions})\nsubmission.to_csv(\"submission.csv\", index=False)\n\nprint(\"Submission file created:\")\nprint(submission.head())\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:39.635563Z","iopub.status.idle":"2024-12-21T02:30:39.635877Z","shell.execute_reply.started":"2024-12-21T02:30:39.635731Z","shell.execute_reply":"2024-12-21T02:30:39.635747Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n<span style=\"font-size:24px;\">Select the best performing model here and submit it</span>","metadata":{}},{"cell_type":"code","source":"\nif xgb_rmsle <= lgb_rmsle and xgb_rmsle <= cat_rmsle:\n    best_model_name = 'XGB'\n    best_predictions = xgb_test_predictions\nelif lgb_rmsle <= xgb_rmsle and lgb_rmsle <= cat_rmsle:\n    best_model_name = 'LGB'\n    best_predictions = lgb_test_predictions\nelse:\n    best_model_name = 'CAT'\n    best_predictions = cat_test_predictions\n\nprint(f\"Best model: {best_model_name}\")\nprint(f\"RMSLE of {best_model_name}: {min(xgb_rmsle, lgb_rmsle, cat_rmsle):.4f}\")\n\ntest_submission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\nsubmission = pd.DataFrame({'id': test_submission['id'], 'Premium Amount': best_predictions})\nsubmission.to_csv(\"submission.csv\", index=False)\n\nprint(\"Submission file created:\")\nprint(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T02:30:39.637405Z","iopub.status.idle":"2024-12-21T02:30:39.637683Z","shell.execute_reply.started":"2024-12-21T02:30:39.63755Z","shell.execute_reply":"2024-12-21T02:30:39.637565Z"}},"outputs":[],"execution_count":null}]}