{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, Input\nfrom tensorflow.keras.layers import Dense, Input, BatchNormalization, Activation, Dropout, Add\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.metrics import RootMeanSquaredError\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom lightgbm import LGBMRegressor\nimport optuna\nfrom sklearn.base import clone\nfrom sklearn.model_selection import KFold\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore')\n\npd.set_option('display.max_rows', None)\npd.set_option('display.max_columns', None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:26:04.755526Z","iopub.execute_input":"2024-12-31T16:26:04.755985Z","iopub.status.idle":"2024-12-31T16:26:18.581291Z","shell.execute_reply.started":"2024-12-31T16:26:04.755947Z","shell.execute_reply":"2024-12-31T16:26:18.579735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ntest_ids = test['id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:26:18.58314Z","iopub.execute_input":"2024-12-31T16:26:18.584423Z","iopub.status.idle":"2024-12-31T16:26:29.794293Z","shell.execute_reply.started":"2024-12-31T16:26:18.584388Z","shell.execute_reply":"2024-12-31T16:26:29.792941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:26:29.796151Z","iopub.execute_input":"2024-12-31T16:26:29.796507Z","iopub.status.idle":"2024-12-31T16:26:29.82626Z","shell.execute_reply.started":"2024-12-31T16:26:29.796478Z","shell.execute_reply":"2024-12-31T16:26:29.824882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def filling_missing_values(data):\n    # Handle missing numerical values with mean\n    data['Age'] = data['Age'].fillna(data['Age'].mean())\n    data['Annual Income'] = data['Annual Income'].fillna(data['Annual Income'].mean())\n    data['Number of Dependents'] = data['Number of Dependents'].fillna(data['Number of Dependents'].mean())\n    data['Health Score'] = data['Health Score'].fillna(data['Health Score'].mean())\n    data['Previous Claims'] = data['Previous Claims'].fillna(data['Previous Claims'].mean())\n    data['Vehicle Age'] = data['Vehicle Age'].fillna(data['Vehicle Age'].mean())\n    data['Credit Score'] = data['Credit Score'].fillna(data['Credit Score'].mean())\n    data['Insurance Duration'] = data['Insurance Duration'].fillna(data['Insurance Duration'].mean())\n    \n    # Handle missing categorical values with mode or special category\n    data['Gender'] = data['Gender'].map({'Male': 0, 'Female': 1}).fillna(-1)  # Use -1 for missing values\n    data['Marital Status'] = data['Marital Status'].fillna(data['Marital Status'].mode().iloc[0])\n    data['Marital Status'] = data['Marital Status'].map({'Single': 0, 'Married': 1, 'Divorced': 2}).fillna(-1)\n    data['Education Level'] = data['Education Level'].map({'PhD': 0, \"Master's\": 1, \"Bachelor's\": 2, \"High School\": 3}).fillna(-1)\n    data['Occupation'] = data['Occupation'].fillna(data['Occupation'].mode().iloc[0])\n    data['Occupation'] = data['Occupation'].map({'Employed': 0, \"Self-Employed\": 1, \"Unemployed\": 2}).fillna(-1)\n    data['Customer Feedback'] = data['Customer Feedback'].fillna(data['Customer Feedback'].mode().iloc[0])\n    data['Customer Feedback'] = data['Customer Feedback'].map({'Good': 0, \"Average\": 1, \"Poor\": 2}).fillna(-1)\n    data['Smoking Status'] = data['Smoking Status'].map({'No': 0, 'Yes': 1}).fillna(-1)\n    data['Exercise Frequency'] = data['Exercise Frequency'].map({'Monthly': 0, 'Weekly': 1, 'Daily': 2, 'Rarely': 3}).fillna(-1)\n    data['Property Type'] = data['Property Type'].map({'House': 0, 'Condo': 1, 'Apartment': 2}).fillna(-1)\n    \n    # Handle missing 'Location' and 'Policy Type'\n    data['Location'] = data['Location'].map({'Suburban': 0, 'Rural': 1, 'Urban': 2}).fillna(-1)\n    data['Policy Type'] = data['Policy Type'].map({'Premium': 0, 'Comprehensive': 1, 'Basic': 2}).fillna(-1)\n\n    # Extract date features from 'Policy Start Date'\n    data['Policy Start Date'] = pd.to_datetime(data['Policy Start Date'], errors='coerce')  # Handle invalid datetime\n    data['policy_year'] = data['Policy Start Date'].dt.year\n    data['policy_month'] = data['Policy Start Date'].dt.month\n    data['policy_day'] = data['Policy Start Date'].dt.day\n    data['policy_hour'] = data['Policy Start Date'].dt.hour\n    data['policy_minute'] = data['Policy Start Date'].dt.minute\n    data['policy_seconds'] = data['Policy Start Date'].dt.second\n    data['policy_day_of_week'] = data['Policy Start Date'].dt.dayofweek\n    data['policy_is_weekend'] = (data['policy_day_of_week'] >= 5).astype(int)\n\n    # Drop original datetime column\n    data.drop(columns=['Policy Start Date'], inplace=True)\n\n    return data\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:26:29.827958Z","iopub.execute_input":"2024-12-31T16:26:29.828397Z","iopub.status.idle":"2024-12-31T16:26:29.842142Z","shell.execute_reply.started":"2024-12-31T16:26:29.828353Z","shell.execute_reply":"2024-12-31T16:26:29.840651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = filling_missing_values(train)\ntest = filling_missing_values(test)\nX = train.drop(columns=['Premium Amount'])\nY = train['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:26:29.843556Z","iopub.execute_input":"2024-12-31T16:26:29.843966Z","iopub.status.idle":"2024-12-31T16:26:34.345634Z","shell.execute_reply.started":"2024-12-31T16:26:29.843927Z","shell.execute_reply":"2024-12-31T16:26:34.344396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:26:34.346827Z","iopub.execute_input":"2024-12-31T16:26:34.347215Z","iopub.status.idle":"2024-12-31T16:26:34.37094Z","shell.execute_reply.started":"2024-12-31T16:26:34.347179Z","shell.execute_reply":"2024-12-31T16:26:34.369654Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Neural Network Model","metadata":{}},{"cell_type":"markdown","source":"> I used a ResNet approach to overcome the vanishing gradient problem, but after reviewing the results, I'm not sure what went wrong. I expected this model to perform very well, but it seems to be doing the opposite. I’ve still included it in the notebook in case anyone has suggestions.","metadata":{}},{"cell_type":"code","source":"def model(input_shape):\n    # Input layer\n    X_input = tf.keras.Input(shape=input_shape)\n    X_shortcut = X_input\n\n    # Layer 1\n    X = layers.Dense(units=64)(X_input)\n    X = layers.BatchNormalization(axis=-1)(X)\n    X = layers.ReLU()(X)\n\n    # Layer 2 with Skip Connection\n    X_shortcut = layers.Dense(units=64)(X_shortcut)  # Match dimensions for skip connection\n    X = layers.Dense(units=64)(X)\n    X = layers.BatchNormalization(axis=-1)(X)\n    X = layers.Add()([X, X_shortcut])  # Adding skip connection\n    X = layers.ReLU()(X)\n\n    # Layer 3\n    X = layers.Dense(units=128)(X)\n    X = layers.BatchNormalization(axis=-1)(X)\n    X = layers.ReLU()(X)\n\n    # Layer 4 with Skip Connection\n    X_shortcut = layers.Dense(units=256)(X)  # Update shortcut for this block\n    X = layers.Dense(units=256)(X)\n    X = layers.BatchNormalization(axis=-1)(X)\n    X = layers.Add()([X, X_shortcut])  # Adding skip connection\n    X = layers.ReLU()(X)\n\n    # Layer 5\n    X = layers.Dense(units=256)(X)\n    X = layers.BatchNormalization(axis=-1)(X)\n    X = layers.ReLU()(X)\n\n    # Output layer\n    output_layer = layers.Dense(units=1, activation=None)(X)\n\n    # Create the model\n    model = tf.keras.Model(inputs=X_input, outputs=output_layer)\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:26:34.372111Z","iopub.execute_input":"2024-12-31T16:26:34.372511Z","iopub.status.idle":"2024-12-31T16:26:34.394957Z","shell.execute_reply.started":"2024-12-31T16:26:34.372472Z","shell.execute_reply":"2024-12-31T16:26:34.393572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train.drop('Premium Amount', axis=1).to_numpy()\nY = train['Premium Amount'].to_numpy()\ntest = test.to_numpy()\n\nX_train, X_val, Y_train, Y_val = train_test_split(X, Y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:26:34.397967Z","iopub.execute_input":"2024-12-31T16:26:34.398336Z","iopub.status.idle":"2024-12-31T16:26:35.246068Z","shell.execute_reply.started":"2024-12-31T16:26:34.398308Z","shell.execute_reply":"2024-12-31T16:26:35.244782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"regression_model = model(input_shape=(X_train.shape[1],))\nregression_model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.001), loss='mean_squared_error', metrics=[RootMeanSquaredError()])\nhistory =regression_model.fit(X_train, Y_train, epochs=20, batch_size=2000, validation_data=(X_val, Y_val))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:26:35.247586Z","iopub.execute_input":"2024-12-31T16:26:35.248078Z","iopub.status.idle":"2024-12-31T16:34:17.013462Z","shell.execute_reply.started":"2024-12-31T16:26:35.248049Z","shell.execute_reply":"2024-12-31T16:34:17.012086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Access RMSE values\ntrain_rmse = history.history['root_mean_squared_error']\nval_rmse = history.history['val_root_mean_squared_error']\n\n# Print final RMSE values\nprint(f\"Final Training RMSE: {train_rmse[-1]:.4f}\")\nprint(f\"Final Validation RMSE: {val_rmse[-1]:.4f}\")\n\n# Plot RMSE over epochs\nplt.plot(train_rmse, label='Training RMSE')\nplt.plot(val_rmse, label='Validation RMSE')\nplt.title('RMSE over Epochs')\nplt.xlabel('Epochs')\nplt.ylabel('RMSE')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:34:17.015347Z","iopub.execute_input":"2024-12-31T16:34:17.015801Z","iopub.status.idle":"2024-12-31T16:34:17.660614Z","shell.execute_reply.started":"2024-12-31T16:34:17.015765Z","shell.execute_reply":"2024-12-31T16:34:17.6592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# prediction = regression_model.predict(test)\n# result =pd.DataFrame ({\n#     'id': test_ids,\n#     'Premium Amount': prediction.flatten()\n# })\n#\n# result.to_csv('Neural_Network5.csv', index=False)\n# LeaderBoard Score: 1.18595\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:34:17.661854Z","iopub.execute_input":"2024-12-31T16:34:17.662204Z","iopub.status.idle":"2024-12-31T16:34:17.666799Z","shell.execute_reply.started":"2024-12-31T16:34:17.66217Z","shell.execute_reply":"2024-12-31T16:34:17.665581Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Using LGBMRegressor Model","metadata":{}},{"cell_type":"code","source":"# def objective(trial):\n#     params = {\n#         'learning_rate': trial.suggest_float('learning_rate', 1e-3, 0.5),\n#         'num_leaves': trial.suggest_int('num_leaves', 150, 1000),\n#         'max_depth': trial.suggest_int('max_depth', 3, 25),\n#         'min_child_samples': trial.suggest_int('min_child_samples', 5, 50),\n#         'subsample': trial.suggest_float('subsample', 0.4, 1.0),\n#         'colsample_bytree': trial.suggest_float('colsample_bytree', 0.4,1.0),\n#         'reg_alpha': trial.suggest_float('reg_alpha', 1e-2, 3),\n#         'reg_lambda': trial.suggest_float('reg_lambda', 1e-2, 3),\n#         # Fixed parameters\n#         'objective': 'regression',\n#         'metric': 'mse',\n#         'boosting_type': 'gbdt',\n#         'verbose': -1\n#     }\n#     kfold = KFold(n_splits=5, shuffle=True, random_state=42)\n#     LGB_model = LGBMRegressor(**params)\n\n#     for train_idx, test_idx in kfold.split(X, Y):\n#         X_train, X_CV = X[train_idx], X[test_idx]\n#         Y_train, Y_CV = Y[train_idx], Y[test_idx]\n\n#         Model = clone(LGB_model)\n#         Model.fit(X_train, Y_train)\n\n#         train_prediction = Model.predict(X_train)\n#         train_error = np.sqrt(mean_squared_error(Y_train, train_prediction))\n#         training_error.append(train_error)\n\n#         prediction = Model.predict(X_CV)\n#         error = np.sqrt(mean_squared_error(Y_CV, prediction))\n#         validation_error.append(error)\n\n#     return np.mean(validation_error)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:34:17.667965Z","iopub.execute_input":"2024-12-31T16:34:17.668248Z","iopub.status.idle":"2024-12-31T16:34:17.684375Z","shell.execute_reply.started":"2024-12-31T16:34:17.668214Z","shell.execute_reply":"2024-12-31T16:34:17.683237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# training_error = []\n# validation_error = []\n# study = optuna.create_study(direction='minimize')\n# study.optimize(objective, n_trials=50)\n# best_parameters = study.best_params\n# print(f'Best Parameters: {best_parameters}')\n\n# plt.plot(training_error, label='Training Error')\n# plt.plot(validation_error, label='Validation Error')\n# plt.title('RMSE Training vs Validation')\n# plt.xlabel('Trials')\n# plt.ylabel('RMSE')\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:34:17.685847Z","iopub.execute_input":"2024-12-31T16:34:17.686171Z","iopub.status.idle":"2024-12-31T16:34:17.707206Z","shell.execute_reply.started":"2024-12-31T16:34:17.686143Z","shell.execute_reply":"2024-12-31T16:34:17.706086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_parameters ={\n                    'learning_rate': 0.05607423932789277,\n                  'num_leaves': 972,\n                  'max_depth': 25,\n                  'min_child_samples': 48,\n                  'subsample': 0.4733368009491471,\n                  'colsample_bytree': 0.4625910542149136,\n                  'reg_alpha': 0.5369303802708364,\n                  'reg_lambda': 1.7172300328737573\n                }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:34:17.708341Z","iopub.execute_input":"2024-12-31T16:34:17.708694Z","iopub.status.idle":"2024-12-31T16:34:17.723413Z","shell.execute_reply.started":"2024-12-31T16:34:17.708657Z","shell.execute_reply":"2024-12-31T16:34:17.721774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Model = LGBMRegressor(**best_parameters, verbosity=-1)\nkfold = KFold(n_splits=5, shuffle=True, random_state=42)\n\nfor train_idx, test_idx in kfold.split(X, Y):\n    X_train, X_CV = X[train_idx], X[test_idx]\n    Y_train, Y_CV = Y[train_idx], Y[test_idx]\n\n    Model.fit(X_train, Y_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:34:17.724821Z","iopub.execute_input":"2024-12-31T16:34:17.725244Z","iopub.status.idle":"2024-12-31T16:35:27.811212Z","shell.execute_reply.started":"2024-12-31T16:34:17.725212Z","shell.execute_reply":"2024-12-31T16:35:27.809927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = Model.predict(test)\nresult = pd.DataFrame({\n    'id': test_ids,\n     'Premium Amount': pred\n})\nresult.to_csv('Predictions.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T16:35:27.812466Z","iopub.execute_input":"2024-12-31T16:35:27.812857Z","iopub.status.idle":"2024-12-31T16:35:35.758438Z","shell.execute_reply.started":"2024-12-31T16:35:27.812787Z","shell.execute_reply":"2024-12-31T16:35:35.757241Z"}},"outputs":[],"execution_count":null}]}