{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:26.186241Z","iopub.execute_input":"2024-12-21T21:41:26.186696Z","iopub.status.idle":"2024-12-21T21:41:26.197851Z","shell.execute_reply.started":"2024-12-21T21:41:26.186663Z","shell.execute_reply":"2024-12-21T21:41:26.196724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Importing essential libraries\nimport numpy as np  # For numerical computations\nimport pandas as pd  # For data manipulation and analysis\nimport matplotlib.pyplot as plt  # For creating static, animated, and interactive visualizations\nimport seaborn as sns  # For statistical data visualization\nimport warnings  # For controlling warning messages\n\n# Suppressing warnings to avoid clutter in output\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:26.902353Z","iopub.execute_input":"2024-12-21T21:41:26.902725Z","iopub.status.idle":"2024-12-21T21:41:26.908179Z","shell.execute_reply.started":"2024-12-21T21:41:26.90269Z","shell.execute_reply":"2024-12-21T21:41:26.907027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reading the dataset from a CSV file\ntrain = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')  # Load the training data from the specified path\ntest_data = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\n\n# Displaying the first few rows of the dataset for inspection\nprint(train.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:27.5964Z","iopub.execute_input":"2024-12-21T21:41:27.596735Z","iopub.status.idle":"2024-12-21T21:41:34.536449Z","shell.execute_reply.started":"2024-12-21T21:41:27.596709Z","shell.execute_reply":"2024-12-21T21:41:34.535229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dropping the 'id' column from the dataset as it's not needed for analysis\ntrain.drop(columns=['id'], inplace=True)  # Remove the 'id' column in-place\ntest_ids = test_data['id']\ntest_data.drop(columns=['id'], inplace=True)\n# Displaying the first few rows of the modified dataset\ntrain.head()  # Show the first 5 rows after the column removal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:34.537816Z","iopub.execute_input":"2024-12-21T21:41:34.538118Z","iopub.status.idle":"2024-12-21T21:41:34.854597Z","shell.execute_reply.started":"2024-12-21T21:41:34.538073Z","shell.execute_reply":"2024-12-21T21:41:34.853489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Checking for missing values in the training dataset\nprint(train.isnull().sum())\nprint(test_data.isnull().sum()) # Returns the count of missing values for each column","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:34.856352Z","iopub.execute_input":"2024-12-21T21:41:34.856664Z","iopub.status.idle":"2024-12-21T21:41:35.908722Z","shell.execute_reply.started":"2024-12-21T21:41:34.856635Z","shell.execute_reply":"2024-12-21T21:41:35.907364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filling missing values in various columns with appropriate strategies\ntrain['Age'].fillna(train['Age'].median(), inplace=True)  # Fill missing 'Age' with the median value\ntrain['Annual Income'].fillna(train['Annual Income'].median(), inplace=True)  # Fill missing 'Annual Income' with the median value\ntrain['Marital Status'].fillna('Unknown', inplace=True)  # Fill missing 'Marital Status' with 'Unknown'\ntrain['Number of Dependents'].fillna(train['Number of Dependents'].median(), inplace=True)  # Fill missing 'Number of Dependents' with the median value\ntrain['Occupation'].fillna('Unknown', inplace=True)  # Fill missing 'Occupation' with 'Unknown'\ntrain['Health Score'].fillna(train['Health Score'].median(), inplace=True)  # Fill missing 'Health Score' with the median value\ntrain['Previous Claims'].fillna(train['Previous Claims'].median(), inplace=True)  # Fill missing 'Previous Claims' with the median value\ntrain['Vehicle Age'].fillna(train['Vehicle Age'].median(), inplace=True)  # Fill missing 'Vehicle Age' with the median value\ntrain['Credit Score'].fillna(train['Credit Score'].median(), inplace=True)  # Fill missing 'Credit Score' with the median value\ntrain['Customer Feedback'].fillna(train['Customer Feedback'].mode()[0], inplace=True)  # Fill missing 'Customer Feedback' with the mode (most frequent value)\ntrain['Insurance Duration'].fillna(train['Insurance Duration'].mode()[0], inplace=True)  # Fill missing 'Insurance Duration' with the mode (most frequent value)\ntrain['Premium Amount'].fillna(train['Premium Amount'].mean(), inplace=True)  # Fill missing 'Premium Amount' with the mean value\ntrain.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:35.909947Z","iopub.execute_input":"2024-12-21T21:41:35.910296Z","iopub.status.idle":"2024-12-21T21:41:37.095223Z","shell.execute_reply.started":"2024-12-21T21:41:35.910266Z","shell.execute_reply":"2024-12-21T21:41:37.093887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handling missing values for test data\ntest_data['Age'].fillna(test_data['Age'].median(), inplace=True)\ntest_data['Annual Income'].fillna(test_data['Annual Income'].median(), inplace=True)\ntest_data['Marital Status'].fillna('Unknown', inplace=True)\ntest_data['Number of Dependents'].fillna(test_data['Number of Dependents'].median(), inplace=True)\ntest_data['Occupation'].fillna('Unknown', inplace=True)\ntest_data['Health Score'].fillna(test_data['Health Score'].median(), inplace=True)\ntest_data['Previous Claims'].fillna(test_data['Previous Claims'].median(), inplace=True)\ntest_data['Vehicle Age'].fillna(test_data['Vehicle Age'].median(), inplace=True)\ntest_data['Credit Score'].fillna(test_data['Credit Score'].median(), inplace=True)\ntest_data['Customer Feedback'].fillna(test_data['Customer Feedback'].mode()[0], inplace=True)\ntest_data['Insurance Duration'].fillna(test_data['Insurance Duration'].mode()[0], inplace=True)\ntest_data.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:37.096457Z","iopub.execute_input":"2024-12-21T21:41:37.096854Z","iopub.status.idle":"2024-12-21T21:41:37.934851Z","shell.execute_reply.started":"2024-12-21T21:41:37.096813Z","shell.execute_reply":"2024-12-21T21:41:37.933814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Displaying the number of unique values for each categorical column\nfor col in train.select_dtypes(include=['object']).columns:  # Loop through all columns with object (string) data type\n    print(f'{col}: {train[col].nunique()}')  # Print the column name and the number of unique values in that column","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:37.935855Z","iopub.execute_input":"2024-12-21T21:41:37.936172Z","iopub.status.idle":"2024-12-21T21:41:39.020914Z","shell.execute_reply.started":"2024-12-21T21:41:37.936145Z","shell.execute_reply":"2024-12-21T21:41:39.019918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Converting 'Policy Start Date' to datetime format, coercing errors to NaT (Not a Time)\ntrain['Policy Start Date'] = pd.to_datetime(train['Policy Start Date'], errors='coerce')  # Ensure proper datetime conversion\n# Extracting individual date-time features from the 'Policy Start Date' column\ntrain['Policy Start Year'] = train['Policy Start Date'].dt.year  # Extract the year\ntrain['Policy Start Month'] = train['Policy Start Date'].dt.month  # Extract the month\n# Drop the original 'Policy Start Date' column as it's no longer needed\ntrain = train.drop(columns=['Policy Start Date'])  # Remove the original date column\n\ntest_data['Policy Start Date'] = pd.to_datetime(test_data['Policy Start Date'], errors='coerce')\ntest_data['Policy Start Year'] = test_data['Policy Start Date'].dt.year\ntest_data['Policy Start Month'] = test_data['Policy Start Date'].dt.month\ntest_data.drop(columns=['Policy Start Date'], inplace=True)\n\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:39.02298Z","iopub.execute_input":"2024-12-21T21:41:39.023353Z","iopub.status.idle":"2024-12-21T21:41:40.19896Z","shell.execute_reply.started":"2024-12-21T21:41:39.023323Z","shell.execute_reply":"2024-12-21T21:41:40.197824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:40.200643Z","iopub.execute_input":"2024-12-21T21:41:40.201062Z","iopub.status.idle":"2024-12-21T21:41:40.223675Z","shell.execute_reply.started":"2024-12-21T21:41:40.201021Z","shell.execute_reply":"2024-12-21T21:41:40.222552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of numerical columns for analysis or model training\nnumerical_cols = [\n    'Age', 'Annual Income', 'Number of Dependents', 'Health Score',\n    'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration',\n    'Policy Start Month', 'Policy Start Year'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:40.224544Z","iopub.execute_input":"2024-12-21T21:41:40.224987Z","iopub.status.idle":"2024-12-21T21:41:40.242887Z","shell.execute_reply.started":"2024-12-21T21:41:40.224951Z","shell.execute_reply":"2024-12-21T21:41:40.241665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import category_encoders as ce\ncategorical_cols= ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n       'Policy Type', 'Customer Feedback', 'Smoking Status',\n       'Exercise Frequency', 'Property Type']\n\nencoder = ce.TargetEncoder()\n\nfor feature in categorical_cols:\n    train[feature] = encoder.fit_transform(train[feature], train['Premium Amount'])\n    test_data[feature] = encoder.transform(test_data[feature])\n    \nprint(\"Columns transformed with target encoder in training data: \")\ncategorical_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:40.243844Z","iopub.execute_input":"2024-12-21T21:41:40.244182Z","iopub.status.idle":"2024-12-21T21:41:50.338931Z","shell.execute_reply.started":"2024-12-21T21:41:40.24415Z","shell.execute_reply":"2024-12-21T21:41:50.337673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Importing StandardScaler for feature scaling\nfrom sklearn.preprocessing import StandardScaler\n\n# Initializing the StandardScaler\nscaler = StandardScaler()\n\n# List of columns to scale (standardize)\ncolumns_to_scale = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score',\n                    'Vehicle Age', 'Credit Score', 'Insurance Duration', 'Policy Start Year', 'Policy Start Month', ]\n\n# Applying standardization to the selected columns\ntrain[columns_to_scale] = scaler.fit_transform(train[columns_to_scale])  # Scaling the features to have mean=0 and variance=1\ntest_data[columns_to_scale] = scaler.transform(test_data[columns_to_scale])  # Scaling the features to have mean=0 and variance=1\n\n# Displaying the first few rows after scaling\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:50.340602Z","iopub.execute_input":"2024-12-21T21:41:50.34095Z","iopub.status.idle":"2024-12-21T21:41:50.640993Z","shell.execute_reply.started":"2024-12-21T21:41:50.340918Z","shell.execute_reply":"2024-12-21T21:41:50.639816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ndef rmsle(true, pred):\n    \"\"\"\n    Calculate the Root Mean Squared Logarithmic Error (RMSLE).\n    \n    Parameters:\n    - true: array-like, true values\n    - pred: array-like, predicted values\n    \n    Returns:\n    - m: float, RMSLE value\n    \"\"\"\n    true_log = np.log1p(true)\n    pred_log = np.log1p(pred)\n    m = np.sqrt(np.mean((true_log - pred_log) ** 2.0))\n    return m\n\n\npred = np.exp(np.mean(np.log1p(train[\"Premium Amount\"]))) - 1\nm = rmsle(train[\"Premium Amount\"].values, pred)\nprint(f\"Exponented Mean Log 1p produces CV RMSLE = {m}\")\n\nX = train.drop([\"Premium Amount\"], axis=1)\ny = train[\"Premium Amount\"]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:41:50.642137Z","iopub.execute_input":"2024-12-21T21:41:50.642418Z","iopub.status.idle":"2024-12-21T21:41:51.447409Z","shell.execute_reply.started":"2024-12-21T21:41:50.642394Z","shell.execute_reply":"2024-12-21T21:41:51.446385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.model_selection import cross_val_predict\n# from sklearn.model_selection import KFold\n\n# def objective(trial):\n    \n#     param = {\n#         'max_depth': trial.suggest_int('max_depth', 9, 13),\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 0.05, 0.1),\n#         'subsample': trial.suggest_uniform('subsample', 0.6, 0.7),\n#         'n_estimators': trial.suggest_int('n_estimators', 90, 110),\n#         'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.6, 0.8),\n#         'gamma': trial.suggest_loguniform('gamma', 1e-3, 1e1),\n#         'min_child_weight': trial.suggest_int('min_child_weight', 1, 5),\n#         'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-3, 1e-1),\n#         'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-3, 1e-1),\n#         'random_state': 42,\n#         'tree_method': 'hist',\n#         'objective': 'reg:gamma',\n#         'verbosity': 0\n#     }\n\n#     # Initialize the model with current trial parameters\n#     model = XGBRegressor(**param)\n    \n#     # Implementing K-Fold Cross-Validation\n#     kf = KFold(n_splits=5, shuffle=True, random_state=42)\n    \n#     # Generate cross-validated predictions\n#     y_preds = cross_val_predict(model, X_train, y_train, cv=kf, method='predict')\n    \n#     # Calculate RMSLE\n#     score = rmsle(y_train, y_preds)\n\n#     return score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:45:51.392164Z","iopub.execute_input":"2024-12-21T19:45:51.392574Z","iopub.status.idle":"2024-12-21T19:45:51.400535Z","shell.execute_reply.started":"2024-12-21T19:45:51.392541Z","shell.execute_reply":"2024-12-21T19:45:51.39945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Create the Optuna study\n# study = optuna.create_study(\n#     direction='minimize',  # We aim to minimize RMSLE\n#     sampler=optuna.samplers.TPESampler(seed=42),\n#     pruner=optuna.pruners.MedianPruner()\n# )\n\n# # Optimize the study\n# study.optimize(objective, n_trials=700, timeout=4500)  # 100 trials or 20 minutes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T20:08:59.278777Z","iopub.execute_input":"2024-12-21T20:08:59.279212Z","iopub.status.idle":"2024-12-21T21:24:31.588709Z","shell.execute_reply.started":"2024-12-21T20:08:59.279177Z","shell.execute_reply":"2024-12-21T21:24:31.587743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Get the best trial\n# best_trial = study.best_trial\n\n# print(\"Best RMSLE: {:.4f}\".format(best_trial.value))\n# print(\"Best hyperparameters:\")\n\n# for key, value in best_trial.params.items():\n#     print(f\"  {key}: {value}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Extract best parameters\n# best_params = study.best_trial.params\n# best_params.update({\n#     'random_state': 42,\n#     'tree_method': 'hist',\n#     'verbosity': 0,\n#     'objective': 'reg:squarederror'  # Ensure consistency\n# })\n\n# # Initialize and train the final model\n# final_model = XGBRegressor(**best_params)\n# final_model.fit(\n#     X_train, y_train,\n#     eval_set=[(X_test, y_test)],\n#     eval_metric='rmse',\n#     early_stopping_rounds=100,\n#     verbose=False\n# )\n\n# # Predict and calculate RMSLE\n# y_pred_final = final_model.predict(X_test)\n# final_rmsle = rmsle(y_test, y_pred_final)\n\n# print(\"\\nFinal RMSLE on validation set:\", final_rmsle)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ndef RMSLE(true, pred):\n    \"\"\"\n    Calculate the Root Mean Squared Logarithmic Error (RMSLE).\n    \n    Parameters:\n    - true: array-like, true values\n    - pred: array-like, predicted values\n    \n    Returns:\n    - m: float, RMSLE value\n    \"\"\"\n    true_log = np.log1p(true)\n    pred_log = np.log1p(pred)\n    m = np.sqrt(np.mean((true_log - pred_log) ** 2.0))\n    return m\n\n\npred = np.exp(np.mean(np.log1p(train[\"Premium Amount\"]))) - 1\nm = RMSLE(train[\"Premium Amount\"].values, pred)\nprint(f\"Exponented Mean Log 1p produces CV RMSLE = {m}\")\n\nX = train.drop([\"Premium Amount\"], axis=1)\ny = train[\"Premium Amount\"]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n\nbest_params = {\n    'max_depth': 11,\n    'learning_rate': 0.06605356819367247,\n    'subsample': 0.6397519686309207,\n    'n_estimators': 100,  \n    'random_state': 42,\n    'tree_method': 'hist',\n    'objective': 'reg:gamma',\n    'verbosity': 0\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:42:37.605031Z","iopub.execute_input":"2024-12-21T21:42:37.605461Z","iopub.status.idle":"2024-12-21T21:42:38.219065Z","shell.execute_reply.started":"2024-12-21T21:42:37.605428Z","shell.execute_reply":"2024-12-21T21:42:38.217991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\nimport optuna\nfrom sklearn.metrics import mean_squared_log_error\n\n# Train XGBRegressor model\nmodel = XGBRegressor(**best_params)\nmodel.fit(\n    X_train, y_train,\n    eval_set=[(X_test, y_test)],  # Validation set for early stopping\n    eval_metric=\"rmse\",            # Use RMSE (Root Mean Squared Error)\n    early_stopping_rounds=200,     # Early stopping\n    verbose=False                  # Suppress logs\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:42:40.84958Z","iopub.execute_input":"2024-12-21T21:42:40.849992Z","iopub.status.idle":"2024-12-21T21:42:56.840853Z","shell.execute_reply.started":"2024-12-21T21:42:40.849962Z","shell.execute_reply":"2024-12-21T21:42:56.839841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict and calculate RMSLE\ny_pred = model.predict(X_test)\nrmsle_score = RMSLE(y_test, y_pred)\n\n# Print the evaluation result\nprint(\"RMSLE on validation set:\", rmsle_score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:42:56.84189Z","iopub.execute_input":"2024-12-21T21:42:56.84222Z","iopub.status.idle":"2024-12-21T21:42:57.296201Z","shell.execute_reply.started":"2024-12-21T21:42:56.842192Z","shell.execute_reply":"2024-12-21T21:42:57.295076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred=model.predict(test_data)\npredictions = np.exp( np.mean( np.log1p(train[\"Premium Amount\"]) ) )-1\n\n# Create a DataFrame to store the ' Id' and corresponding 'Premium Amount' predictions\noutput = pd.DataFrame({'id': test_ids, 'Premium Amount': predictions})\n\n# Save the predictions to a CSV file for submission\noutput.to_csv('submission.csv', index=False)\n\nprint(\"Submission was successfully saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T21:42:57.298041Z","iopub.execute_input":"2024-12-21T21:42:57.298516Z","iopub.status.idle":"2024-12-21T21:43:00.328171Z","shell.execute_reply.started":"2024-12-21T21:42:57.298466Z","shell.execute_reply":"2024-12-21T21:43:00.326876Z"}},"outputs":[],"execution_count":null}]}