{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":10256213,"sourceType":"datasetVersion","datasetId":6344508}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:05:09.734749Z","iopub.execute_input":"2024-12-22T17:05:09.735056Z","iopub.status.idle":"2024-12-22T17:05:10.028772Z","shell.execute_reply.started":"2024-12-22T17:05:09.735027Z","shell.execute_reply":"2024-12-22T17:05:10.028114Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Importing Necessary Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, regularizers\nfrom tensorflow.keras import backend as K\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport gc  # For memory management\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.optimizers import Adam","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:05:39.79843Z","iopub.execute_input":"2024-12-22T17:05:39.798722Z","iopub.status.idle":"2024-12-22T17:05:48.252348Z","shell.execute_reply.started":"2024-12-22T17:05:39.798699Z","shell.execute_reply":"2024-12-22T17:05:48.251218Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Defining the RMSLE Loss Function","metadata":{}},{"cell_type":"code","source":"def rmsle(y_true, y_pred):\n    \"\"\"\n    Root Mean Squared Logarithmic Error (RMSLE) loss function.\n    \"\"\"\n    epsilon = K.epsilon()\n    y_pred = K.clip(y_pred, epsilon, None)  # Prevent negative predictions\n    y_true = K.clip(y_true, epsilon, None)  # Prevent negative true values\n    return K.sqrt(K.mean(K.square(K.log(1 + y_pred) - K.log(1 + y_true))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:05:48.253522Z","iopub.execute_input":"2024-12-22T17:05:48.253998Z","iopub.status.idle":"2024-12-22T17:05:48.258638Z","shell.execute_reply.started":"2024-12-22T17:05:48.253974Z","shell.execute_reply":"2024-12-22T17:05:48.257719Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Loading and Exploring Datasets","metadata":{}},{"cell_type":"code","source":"# Define paths to train and test datasets\ntrain_file_path = '/kaggle/input/3-and-4/data_train_3.csv'  # Path to the training dataset\ntest_file_path = '/kaggle/input/3-and-4/data_test_3.csv'    # Path to the test dataset\n\n# Name of the target column\ntarget_column = 'Premium Amount'\n\n# Load the training dataset\nprint(\"Loading training dataset...\")\ntrain_df = pd.read_csv(train_file_path, dtype={'Gender': 'int64', 'Smoking Status': 'int64'})\nprint(f\"Training dataset size: {train_df.shape}\")\n\n# Load the test dataset\nprint(\"Loading test dataset...\")\ntest_df = pd.read_csv(test_file_path, dtype={'Gender': 'int64', 'Smoking Status': 'int64'})\nprint(f\"Test dataset size: {test_df.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:05:48.260285Z","iopub.execute_input":"2024-12-22T17:05:48.260481Z","iopub.status.idle":"2024-12-22T17:06:13.800429Z","shell.execute_reply.started":"2024-12-22T17:05:48.260463Z","shell.execute_reply":"2024-12-22T17:06:13.799664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the number of columns in the datasets\nprint(f\"\\nNumber of columns in the training dataset (excluding the target column): {train_df.shape[1] - 1}\")\nprint(f\"Number of columns in the test dataset: {test_df.shape[1]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:06:13.801642Z","iopub.execute_input":"2024-12-22T17:06:13.801947Z","iopub.status.idle":"2024-12-22T17:06:13.806196Z","shell.execute_reply.started":"2024-12-22T17:06:13.801924Z","shell.execute_reply":"2024-12-22T17:06:13.805515Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Data Preprocessing","metadata":{}},{"cell_type":"code","source":"# Separate features and target variable\nX = train_df.drop(columns=[target_column]).values\ny = train_df[target_column].values.reshape(-1, 1)\n\n# Separate features in the test dataset (excluding the 'id' column)\nX_test = test_df.drop(columns=['id']).values\ntest_ids = test_df['id'].values  # Save the IDs for predictions\n\nprint(f\"\\nFeature matrix shape (X): {X.shape}\")\nprint(f\"Target variable shape (y): {y.shape}\")\nprint(f\"Test dataset feature matrix shape (X_test): {X_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:06:13.836702Z","iopub.execute_input":"2024-12-22T17:06:13.836958Z","iopub.status.idle":"2024-12-22T17:06:17.158952Z","shell.execute_reply.started":"2024-12-22T17:06:13.836931Z","shell.execute_reply":"2024-12-22T17:06:17.157999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data Normalization\nprint(\"\\nPerforming data normalization...\")\nscaler = StandardScaler()\nX = scaler.fit_transform(X)\nX_test = scaler.transform(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:06:17.15984Z","iopub.execute_input":"2024-12-22T17:06:17.160177Z","iopub.status.idle":"2024-12-22T17:06:25.119962Z","shell.execute_reply.started":"2024-12-22T17:06:17.160153Z","shell.execute_reply":"2024-12-22T17:06:25.119263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Splitting the dataset into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(\n    X, y, test_size=0.1, random_state=42\n)\nprint(f\"\\nTraining set size: {X_train.shape}\")\nprint(f\"Validation set size: {X_val.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:06:25.121994Z","iopub.execute_input":"2024-12-22T17:06:25.122249Z","iopub.status.idle":"2024-12-22T17:06:26.009073Z","shell.execute_reply.started":"2024-12-22T17:06:25.122226Z","shell.execute_reply":"2024-12-22T17:06:26.008257Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Building the Model","metadata":{}},{"cell_type":"code","source":"def build_model_with_bn_new(input_dim):\n    \"\"\"\n    Creates a regression model with Batch Normalization and uses RMSLE as the loss function.\n    \n    :param input_dim: Input dimension.\n    :return: Compiled model.\n    \"\"\"\n    model_new = models.Sequential()\n    model_new.add(layers.Dense(512, activation='relu', input_shape=(input_dim,),\n                               kernel_regularizer=regularizers.l2(0.001)))\n    model_new.add(layers.BatchNormalization())\n    model_new.add(layers.Dropout(0.5))\n    \n    model_new.add(layers.Dense(256, activation='relu',\n                                kernel_regularizer=regularizers.l2(0.001)))\n    model_new.add(layers.BatchNormalization())\n    model_new.add(layers.Dropout(0.9))\n    \n    model_new.add(layers.Dense(256, activation='relu',\n                                kernel_regularizer=regularizers.l2(0.001)))\n    model_new.add(layers.BatchNormalization())\n    model_new.add(layers.Dropout(0.9))\n\n    model_new.add(layers.Dense(256, activation='relu',\n                                kernel_regularizer=regularizers.l2(0.001)))\n    model_new.add(layers.BatchNormalization())\n    model_new.add(layers.Dropout(0.5))\n    \n    model_new.add(layers.Dense(1, activation='linear'))\n    \n    # Define the optimizer with a specific learning rate\n    optimizer = Adam(learning_rate=0.00001)\n    \n    # Compile the model using RMSLE as the loss function\n    model_new.compile(optimizer=optimizer, loss=rmsle, metrics=[rmsle, 'mae'])\n    return model_new","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:06:26.010087Z","iopub.execute_input":"2024-12-22T17:06:26.010403Z","iopub.status.idle":"2024-12-22T17:06:26.016783Z","shell.execute_reply.started":"2024-12-22T17:06:26.01038Z","shell.execute_reply":"2024-12-22T17:06:26.015856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the input dimension\ninput_dim = X_train.shape[1]\n\n# Build the new model\nmodel_new = build_model_with_bn_new(input_dim)\nmodel_new.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:06:26.017669Z","iopub.execute_input":"2024-12-22T17:06:26.017873Z","iopub.status.idle":"2024-12-22T17:06:27.140242Z","shell.execute_reply.started":"2024-12-22T17:06:26.017856Z","shell.execute_reply":"2024-12-22T17:06:27.139411Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Training the Model","metadata":{}},{"cell_type":"code","source":"# Define Early Stopping and Model Checkpoint Callbacks\nearly_stop_new = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\ncheckpoint_new = ModelCheckpoint('best_model_new.keras', monitor='val_loss', save_best_only=True)\n\n# Train the new model\nprint(\"\\nTraining the new model...\")\nhistory_new = model_new.fit(\n    X_train, y_train,\n    epochs=200,  # Higher epoch count, controlled by early stopping\n    batch_size=256,\n    validation_data=(X_val, y_val),\n    callbacks=[early_stop_new, checkpoint_new],\n    verbose=1\n)\nprint(\"Training completed.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:06:27.141073Z","iopub.execute_input":"2024-12-22T17:06:27.141316Z","iopub.status.idle":"2024-12-22T17:19:09.551626Z","shell.execute_reply.started":"2024-12-22T17:06:27.141286Z","shell.execute_reply":"2024-12-22T17:19:09.550861Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7. Evaluating the Model","metadata":{}},{"cell_type":"code","source":"# Evaluate the model\nprint(\"\\nEvaluating the model...\")\nloss, rmsle_metric, mae = model_new.evaluate(X_val, y_val, verbose=1)\nprint(f\"Validation Loss (RMSLE): {loss}\")\nprint(f\"Validation RMSLE Metric: {rmsle_metric}\")\nprint(f\"Validation MAE: {mae}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:19:09.552432Z","iopub.execute_input":"2024-12-22T17:19:09.552731Z","iopub.status.idle":"2024-12-22T17:19:15.766872Z","shell.execute_reply.started":"2024-12-22T17:19:09.552707Z","shell.execute_reply":"2024-12-22T17:19:15.766048Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 8. Making Predictions on the Test Set and Saving Results","metadata":{}},{"cell_type":"code","source":"# Make predictions on the test set\nprint(\"\\nMaking predictions on the test set...\")\npredictions = model_new.predict(X_test).flatten()\n\n\n# Save the results\nsubmission = pd.DataFrame({\n    'id': test_ids,\n    'Premium Amount': predictions\n})\nsubmission.to_csv('submission_nn_kg2.csv', index=False)\nprint(\"Predictions saved to 'submission.csv'.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T17:19:15.767808Z","iopub.execute_input":"2024-12-22T17:19:15.768154Z","iopub.status.idle":"2024-12-22T17:20:01.671627Z","shell.execute_reply.started":"2024-12-22T17:19:15.768122Z","shell.execute_reply":"2024-12-22T17:20:01.670813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}