{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd \nimport numpy as np \nimport matplotlib.pyplot as plt\nplt.style.use('ggplot')\nimport seaborn as sns\n\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:52:59.444598Z","iopub.execute_input":"2024-12-01T18:52:59.444993Z","iopub.status.idle":"2024-12-01T18:52:59.451731Z","shell.execute_reply.started":"2024-12-01T18:52:59.444962Z","shell.execute_reply":"2024-12-01T18:52:59.450353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler, StandardScaler, LabelEncoder\nfrom sklearn.model_selection import KFold, StratifiedKFold, train_test_split, GridSearchCV, RepeatedKFold, RepeatedStratifiedKFold\nfrom sklearn.ensemble import RandomForestRegressor, HistGradientBoostingRegressor, GradientBoostingRegressor, ExtraTreesRegressor\nfrom sklearn.svm import SVR","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:53:01.062609Z","iopub.execute_input":"2024-12-01T18:53:01.063042Z","iopub.status.idle":"2024-12-01T18:53:01.069176Z","shell.execute_reply.started":"2024-12-01T18:53:01.063002Z","shell.execute_reply":"2024-12-01T18:53:01.067973Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Finding out if i Imported Correctly ","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\n\nprint(train.shape)\nprint(test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:41:55.296216Z","iopub.execute_input":"2024-12-01T17:41:55.296677Z","iopub.status.idle":"2024-12-01T17:42:05.857208Z","shell.execute_reply.started":"2024-12-01T17:41:55.296639Z","shell.execute_reply":"2024-12-01T17:42:05.856112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Personal prefrence but using sample is better than head and tail as it gives a wider approach\n\ntrain.sample(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:43:30.198721Z","iopub.execute_input":"2024-12-01T17:43:30.199108Z","iopub.status.idle":"2024-12-01T17:43:30.266764Z","shell.execute_reply.started":"2024-12-01T17:43:30.199077Z","shell.execute_reply":"2024-12-01T17:43:30.265329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.sample(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:42:50.197628Z","iopub.execute_input":"2024-12-01T17:42:50.198029Z","iopub.status.idle":"2024-12-01T17:42:50.247962Z","shell.execute_reply.started":"2024-12-01T17:42:50.197988Z","shell.execute_reply":"2024-12-01T17:42:50.246595Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Science 101 -> Missing values ","metadata":{}},{"cell_type":"code","source":"# Got to know about this from another code on kaggle shoutout to him for making me remember percentage is a better metric \n\nprint(100*train.isnull().sum() / train.shape[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:46:44.15291Z","iopub.execute_input":"2024-12-01T17:46:44.153303Z","iopub.status.idle":"2024-12-01T17:46:44.798777Z","shell.execute_reply.started":"2024-12-01T17:46:44.153268Z","shell.execute_reply":"2024-12-01T17:46:44.797557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(100*test.isnull().sum() / test.shape[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:46:53.829954Z","iopub.execute_input":"2024-12-01T17:46:53.830335Z","iopub.status.idle":"2024-12-01T17:46:54.260957Z","shell.execute_reply.started":"2024-12-01T17:46:53.830304Z","shell.execute_reply":"2024-12-01T17:46:54.259843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Useless because kaggle is not gonna make this basic mistake but better safe than sorry \n\nprint (train.duplicated().sum())\nprint (test.duplicated().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:48:42.743795Z","iopub.execute_input":"2024-12-01T17:48:42.744185Z","iopub.status.idle":"2024-12-01T17:48:45.725612Z","shell.execute_reply.started":"2024-12-01T17:48:42.74415Z","shell.execute_reply":"2024-12-01T17:48:45.724557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.info())\nprint(test.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:53:49.303238Z","iopub.execute_input":"2024-12-01T17:53:49.304745Z","iopub.status.idle":"2024-12-01T17:53:50.438027Z","shell.execute_reply.started":"2024-12-01T17:53:49.304671Z","shell.execute_reply":"2024-12-01T17:53:50.436794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.describe())\nprint(test.describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:54:15.506533Z","iopub.execute_input":"2024-12-01T17:54:15.506927Z","iopub.status.idle":"2024-12-01T17:54:16.651096Z","shell.execute_reply.started":"2024-12-01T17:54:15.506891Z","shell.execute_reply":"2024-12-01T17:54:16.649832Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Lets add some color and style ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(12, 6))\n\nplt_1 = sns.scatterplot(data=train, x='Age', y='Premium Amount', ax = ax[0])\nplt_2 = sns.boxplot(data=train, x='Gender', y='Premium Amount', ax = ax[1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:51:12.700093Z","iopub.execute_input":"2024-12-01T17:51:12.700607Z","iopub.status.idle":"2024-12-01T17:51:18.396849Z","shell.execute_reply.started":"2024-12-01T17:51:12.700556Z","shell.execute_reply":"2024-12-01T17:51:18.39531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 7))\n\nplt_1 = sns.boxplot(data=train, x='Previous Claims', y='Premium Amount', ax=ax[0])\nplt_2 = sns.boxplot(data=train, x='Policy Type', y='Premium Amount', ax=ax[1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:52:14.870517Z","iopub.execute_input":"2024-12-01T17:52:14.8714Z","iopub.status.idle":"2024-12-01T17:52:16.322203Z","shell.execute_reply.started":"2024-12-01T17:52:14.871356Z","shell.execute_reply":"2024-12-01T17:52:16.320952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# doing for previous claims because it felt like i could find something totally not my intention add color and waste my time \n\nsns.pairplot(train.sample(frac=0.03, random_state=42))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:05:09.265553Z","iopub.execute_input":"2024-12-01T18:05:09.265964Z","iopub.status.idle":"2024-12-01T18:05:50.410646Z","shell.execute_reply.started":"2024-12-01T18:05:09.265927Z","shell.execute_reply":"2024-12-01T18:05:50.409388Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# The irritaing part","metadata":{}},{"cell_type":"code","source":"def feature_processing(df):\n\n    df['Gender'] = df['Gender'].map({'Female': 0, 'Male': 1})\n    df['Smoking Status'] = df['Smoking Status'].map({'No': 0, 'Yes': 1})\n    df['Previous Claims'] = df['Previous Claims'].clip(None, 8)    \n\n    return df\n\ntrain = feature_processing(train)\ntest = feature_processing(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:07:17.348843Z","iopub.execute_input":"2024-12-01T18:07:17.349274Z","iopub.status.idle":"2024-12-01T18:07:17.668993Z","shell.execute_reply.started":"2024-12-01T18:07:17.349236Z","shell.execute_reply":"2024-12-01T18:07:17.667985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sorry the best i could do was this next time will try harder \n\ncat_cols = ['Marital Status',\n 'Education Level',\n 'Occupation',\n 'Location',\n 'Policy Type',\n 'Customer Feedback',\n 'Exercise Frequency',\n 'Property Type']\n\ntrain = train.drop(columns=cat_cols)\ntest = test.drop(columns=cat_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:12:49.388437Z","iopub.execute_input":"2024-12-01T18:12:49.388965Z","iopub.status.idle":"2024-12-01T18:12:49.639725Z","shell.execute_reply.started":"2024-12-01T18:12:49.388908Z","shell.execute_reply":"2024-12-01T18:12:49.638594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train.drop(columns=['Policy Start Date', 'Premium Amount'], axis=1)\ny = train['Premium Amount']\n\nskf = RepeatedKFold(n_splits=2, n_repeats=1, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:53:26.753408Z","iopub.execute_input":"2024-12-01T18:53:26.754246Z","iopub.status.idle":"2024-12-01T18:53:27.056822Z","shell.execute_reply.started":"2024-12-01T18:53:26.754202Z","shell.execute_reply":"2024-12-01T18:53:27.055493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"This custom mean_squared_log_error function calculates the Mean Squared Logarithmic Error (MSLE), often used when predicting targets like prices or counts where values vary significantly and we want to penalize relative differences rather than absolute errors.\n\nnp.log1p: Applies log(1 + x) to avoid issues with small or zero values in y_true and y_pred.\n\nWhy use MSLE? It focuses on the percentage differences, making it suitable when large predictions are more tolerable than large absolute errors in small values.\"\"\"\n\ndef mean_squared_log_error(y_true, y_pred):\n    return np.mean((np.log1p(y_true) - np.log1p(y_pred)) ** 2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:53:30.840224Z","iopub.execute_input":"2024-12-01T18:53:30.840657Z","iopub.status.idle":"2024-12-01T18:53:30.847426Z","shell.execute_reply.started":"2024-12-01T18:53:30.840608Z","shell.execute_reply":"2024-12-01T18:53:30.846022Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Honestly Biggest Take away pls never give 10 folds 5 is enough the coumutation time is seriously irritating","metadata":{}},{"cell_type":"code","source":"from catboost import Pool, CatBoostRegressor\n\nscores = []\ncat_test_preds = []  # Initialize for storing final test predictions\n\nfor i, (train_index, test_index) in enumerate(skf.split(X, y)):\n    X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n    y_train, y_test = y[train_index], y[test_index]\n\n    # Create CatBoost Pools for training and validation\n    model_pool = Pool(data=X_train, label=y_train)\n    eval_pool = Pool(data=X_test, label=y_test)\n    \n    # Define the model and fit\n    cat_r = CatBoostRegressor()  # Instantiate the model\n    cat_r.fit(model_pool, eval_set=eval_pool, verbose=0)\n    \n    # Predict and calculate RMSLE score\n    preds = cat_r.predict(eval_pool)\n    score = mean_squared_log_error(y_test, preds)\n    print(f\"The OOF RMSLE score is {score}\")\n    scores.append(score)\n\n\n    # Make predictions on the test set and append them\n    # Here, we use the same `cat_r` model trained in each fold\n    preds_test = cat_r.predict(X_test)  # Use X_test for the predictions\n    cat_test_preds.append(preds_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T18:32:32.212757Z","iopub.execute_input":"2024-12-01T18:32:32.213202Z","iopub.status.idle":"2024-12-01T18:45:32.599433Z","shell.execute_reply.started":"2024-12-01T18:32:32.213166Z","shell.execute_reply":"2024-12-01T18:45:32.597789Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# you may be asking why 2 models the above one was comming with a value error therefore i had to work my way around it \n\nthis notebook is not the optimal solution but my learning journey and therefore feedbacks are appreticated if someone wants to tell me where i went wrong pls do that and for others who found it helpful just enjoy making your own and belive because going wrong really really messes it up\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import mean_squared_log_error\n\n# Define the feature processing function\ndef feature_processing(df):\n    # Map gender to numeric values\n    df['Gender'] = df['Gender'].map({'Female': 0, 'Male': 1})\n    \n    # Map smoking status to numeric values\n    df['Smoking Status'] = df['Smoking Status'].map({'No': 0, 'Yes': 1})\n    \n    # Clip the 'Previous Claims' column, capping values at 8\n    df['Previous Claims'] = df['Previous Claims'].clip(None, 8)\n    \n    return df\n\n# List of categorical columns to drop\ncat_cols = ['Marital Status', 'Education Level', 'Occupation', 'Location', \n            'Policy Type', 'Customer Feedback', 'Exercise Frequency', 'Property Type']\n\n# Load the data\ntrain = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv') \ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')   \n\n# Drop categorical columns from train and test datasets\ntrain = train.drop(columns=cat_cols)\ntest = test.drop(columns=cat_cols)\n\n# Apply the feature processing function to both train and test datasets\ntrain = feature_processing(train)\ntest = feature_processing(test)\n\n# Drop 'Policy Start Date' column\ntrain = train.drop(columns=['Policy Start Date'], axis=1)\ntest = test.drop(columns=['Policy Start Date'], axis=1)\n\n# Define the target and feature columns\nX = train.drop(columns=['Premium Amount'], axis=1)\ny = train['Premium Amount']\n\n# Initialize StratifiedKFold\nskf = StratifiedKFold(n_splits=3, shuffle=True, random_state=42)\n\n# Initialize lists to store predictions\nscores = []\ncat_test_preds = []  # List to store predictions for the entire test set\n\n# Cross-validation loop\nfor i, (train_index, test_index) in enumerate(skf.split(X, y)):\n    X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n    y_train, y_test = y[train_index], y[test_index]\n\n    # Create Pool for CatBoost model\n    model_pool = Pool(data=X_train, label=y_train)\n    eval_pool = Pool(data=X_test, label=y_test)\n\n    # Initialize CatBoost model\n    cat_r = CatBoostRegressor(iterations=1000, learning_rate=0.05, depth=10, loss_function='RMSE', verbose=0)\n\n    # Fit the model\n    cat_r.fit(model_pool, eval_set=eval_pool, verbose=0)\n\n    # Make predictions on the test set (out-of-fold predictions)\n    preds = cat_r.predict(eval_pool)\n\n    # Calculate RMSLE (Root Mean Squared Logarithmic Error)\n    score = mean_squared_log_error(y_test, preds)\n    print(f\"Fold {i + 1}: The oof RMSLE score is {score}\")\n    scores.append(score)\n\n    # Collect predictions for the full test set\n    cat_test_preds.append(cat_r.predict(Pool(data=test)))\n\n# Average out the predictions from cross-validation for the full test set\nfinal_predictions = np.mean(cat_test_preds, axis=0)\n\n# Ensure the length of final_predictions matches the test set length\nassert len(final_predictions) == len(test), f\"Length mismatch: {len(final_predictions)} vs {len(test)}\"\n\n\n\n# Output the final RMSLE score across all folds\naverage_score = np.mean(scores)\nprint(f'Average RMSLE score across all folds: {average_score}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T19:12:08.911483Z","iopub.execute_input":"2024-12-01T19:12:08.911938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare the submission DataFrame\nsubmission = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')  \n\n# Update the submission with final predictions\nsubmission['Premium Amount'] = final_predictions\n\n# Save the submission file\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T19:19:34.551408Z","iopub.execute_input":"2024-12-01T19:19:34.551862Z","iopub.status.idle":"2024-12-01T19:19:36.463857Z","shell.execute_reply.started":"2024-12-01T19:19:34.551822Z","shell.execute_reply":"2024-12-01T19:19:36.462561Z"}},"outputs":[],"execution_count":null}]}