{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The links to my previous notebooks:\n1. EDA Notebook: [EDA](http://www.kaggle.com/code/baibhavkundu2005/insurance-regression-eda). It contains visual insights which will help you to decide how to proceed with imputation and transformations.\n2. Preprocessing + AutoML: [AutoML](http://www.kaggle.com/code/baibhavkundu2005/insurance-preprocessing-automl).Data preprocessing steps with basic AutoML.\n\nIf you like them, do upvote.Thank you!\n\n","metadata":{}},{"cell_type":"markdown","source":"This notebook was taking too much time to be submitted hence I checked my score by directly submitting the \"submission.csv\" file. I hope this notebook is helpful to you.","metadata":{}},{"cell_type":"markdown","source":"# In this notebook, I have implemented a Neural Network for regression purpose and to compare its performance with other ML algorithms.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-03T11:01:54.055484Z","iopub.execute_input":"2024-12-03T11:01:54.055975Z","iopub.status.idle":"2024-12-03T11:01:54.561226Z","shell.execute_reply.started":"2024-12-03T11:01:54.055921Z","shell.execute_reply":"2024-12-03T11:01:54.559937Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Importing libraries and modules","metadata":{}},{"cell_type":"code","source":"from sklearn.experimental import enable_iterative_imputer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:01:54.562553Z","iopub.execute_input":"2024-12-03T11:01:54.563097Z","iopub.status.idle":"2024-12-03T11:01:55.174406Z","shell.execute_reply.started":"2024-12-03T11:01:54.563059Z","shell.execute_reply":"2024-12-03T11:01:55.173449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:01:55.176247Z","iopub.execute_input":"2024-12-03T11:01:55.176893Z","iopub.status.idle":"2024-12-03T11:01:57.360868Z","shell.execute_reply.started":"2024-12-03T11:01:55.176848Z","shell.execute_reply":"2024-12-03T11:01:57.359298Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**To ensure reproducibility**","metadata":{}},{"cell_type":"code","source":"torch.manual_seed(42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:01:57.362377Z","iopub.execute_input":"2024-12-03T11:01:57.363046Z","iopub.status.idle":"2024-12-03T11:01:57.374826Z","shell.execute_reply.started":"2024-12-03T11:01:57.363001Z","shell.execute_reply":"2024-12-03T11:01:57.373493Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Importing files","metadata":{}},{"cell_type":"code","source":"df_tr = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_ts = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ndf_s = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:01:57.379525Z","iopub.execute_input":"2024-12-03T11:01:57.380766Z","iopub.status.idle":"2024-12-03T11:02:07.934271Z","shell.execute_reply.started":"2024-12-03T11:01:57.380674Z","shell.execute_reply":"2024-12-03T11:02:07.933187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import IterativeImputer\nfrom sklearn.impute import SimpleImputer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:02:07.935734Z","iopub.execute_input":"2024-12-03T11:02:07.93617Z","iopub.status.idle":"2024-12-03T11:02:07.941749Z","shell.execute_reply.started":"2024-12-03T11:02:07.936132Z","shell.execute_reply":"2024-12-03T11:02:07.94044Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing: Imputation and Encoding","metadata":{}},{"cell_type":"code","source":"def fill_nulls(train_df, test_df, target_col):\n    X_train = train_df.drop(columns=[target_col])\n    y_train = train_df[target_col]\n\n    numeric_cols = X_train.select_dtypes(include=['float64', 'int64']).columns\n    categorical_cols = X_train.select_dtypes(include=['object', 'category']).columns\n\n    # Numeric imputation\n    iterative_imputer = IterativeImputer(max_iter=10, random_state=42)\n    X_train[numeric_cols] = iterative_imputer.fit_transform(X_train[numeric_cols])\n    test_df[numeric_cols] = iterative_imputer.transform(test_df[numeric_cols])\n\n    # Categorical imputation\n    categorical_imputer = SimpleImputer(strategy='constant', fill_value='Unknown')\n    X_train[categorical_cols] = categorical_imputer.fit_transform(X_train[categorical_cols])\n    test_df[categorical_cols] = categorical_imputer.transform(test_df[categorical_cols])\n\n    train_df_filled = X_train.copy()\n    train_df_filled[target_col] = y_train\n\n    return train_df_filled, test_df\n\ndf_tr_filled, df_ts_filled = fill_nulls(df_tr, df_ts, target_col='Premium Amount')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:02:07.943295Z","iopub.execute_input":"2024-12-03T11:02:07.943803Z","iopub.status.idle":"2024-12-03T11:02:39.87837Z","shell.execute_reply.started":"2024-12-03T11:02:07.943761Z","shell.execute_reply":"2024-12-03T11:02:39.877264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def encode_categorical(train_df, test_df):\n    label_encoders = {}\n    for col in train_df.select_dtypes(include=['object', 'category']).columns:\n        le = LabelEncoder()\n        train_df[col] = le.fit_transform(train_df[col])\n        \n        # Extend the classes to include 'unknown'\n        classes = list(le.classes_)\n        classes.append('unknown')\n        le.classes_ = np.array(classes)\n        \n        # Vectorized transformation for the test set\n        test_df[col] = test_df[col].apply(lambda x: x if x in le.classes_ else 'unknown')\n        test_df[col] = le.transform(test_df[col])\n        \n        label_encoders[col] = le\n    return train_df, test_df, label_encoders","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:02:39.879922Z","iopub.execute_input":"2024-12-03T11:02:39.880306Z","iopub.status.idle":"2024-12-03T11:02:39.888452Z","shell.execute_reply.started":"2024-12-03T11:02:39.880268Z","shell.execute_reply":"2024-12-03T11:02:39.887013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_tr_encoded, df_ts_encoded, encoders = encode_categorical(df_tr_filled, df_ts_filled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:11:19.461214Z","iopub.execute_input":"2024-12-03T11:11:19.461682Z","iopub.status.idle":"2024-12-03T11:11:34.408319Z","shell.execute_reply.started":"2024-12-03T11:11:19.461646Z","shell.execute_reply":"2024-12-03T11:11:34.407132Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preparing X and y for splitting","metadata":{}},{"cell_type":"code","source":"# Prepare features and target\nX = df_tr_encoded.drop(columns=['Premium Amount'])\ny = df_tr_encoded['Premium Amount']\n\n# Apply log transformation to the target\ny_log = np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:11:44.503774Z","iopub.execute_input":"2024-12-03T11:11:44.504808Z","iopub.status.idle":"2024-12-03T11:11:44.640795Z","shell.execute_reply.started":"2024-12-03T11:11:44.504748Z","shell.execute_reply":"2024-12-03T11:11:44.639434Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Converting into PyTorch tensors**","metadata":{}},{"cell_type":"code","source":"X_tensor = torch.tensor(X.values, dtype=torch.float32)\ny_tensor = torch.tensor(y_log.values, dtype=torch.float32).view(-1, 1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:11:47.104129Z","iopub.execute_input":"2024-12-03T11:11:47.10462Z","iopub.status.idle":"2024-12-03T11:11:47.288243Z","shell.execute_reply.started":"2024-12-03T11:11:47.104577Z","shell.execute_reply":"2024-12-03T11:11:47.286836Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train-Test Split","metadata":{}},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X_tensor, y_tensor, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:11:49.620953Z","iopub.execute_input":"2024-12-03T11:11:49.622313Z","iopub.status.idle":"2024-12-03T11:11:50.316769Z","shell.execute_reply.started":"2024-12-03T11:11:49.622261Z","shell.execute_reply":"2024-12-03T11:11:50.315484Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Creating a tensor dataset for our PyTorch based neural network**","metadata":{}},{"cell_type":"code","source":"train_dataset = TensorDataset(X_train, y_train)\nval_dataset = TensorDataset(X_val, y_val)\n\ntrain_loader = DataLoader(train_dataset, batch_size=64, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=64, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:11:52.16728Z","iopub.execute_input":"2024-12-03T11:11:52.167965Z","iopub.status.idle":"2024-12-03T11:11:52.176485Z","shell.execute_reply.started":"2024-12-03T11:11:52.167905Z","shell.execute_reply":"2024-12-03T11:11:52.175051Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Setting up the neural network","metadata":{}},{"cell_type":"code","source":"class RegressionNN(nn.Module):\n    def __init__(self, input_dim):\n        super(RegressionNN, self).__init__()\n        self.network = nn.Sequential(\n            nn.Linear(input_dim, 128),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(128, 64),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(64, 1)\n        )\n\n    def forward(self, x):\n        return self.network(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:11:56.153863Z","iopub.execute_input":"2024-12-03T11:11:56.154352Z","iopub.status.idle":"2024-12-03T11:11:56.162206Z","shell.execute_reply.started":"2024-12-03T11:11:56.154312Z","shell.execute_reply":"2024-12-03T11:11:56.160775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = RegressionNN(input_dim=X_train.shape[1]).to(device)\ncriterion = nn.MSELoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001, weight_decay=1e-5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:11:58.862907Z","iopub.execute_input":"2024-12-03T11:11:58.863391Z","iopub.status.idle":"2024-12-03T11:11:59.854228Z","shell.execute_reply.started":"2024-12-03T11:11:58.863352Z","shell.execute_reply":"2024-12-03T11:11:59.852753Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Custom function for Root Mean Squared Logarithmic Error**","metadata":{}},{"cell_type":"code","source":"def calculate_rmsle(y_true, y_pred):\n    y_true = torch.expm1(y_true)\n    y_pred = torch.expm1(y_pred)\n    return torch.sqrt(torch.mean((torch.log1p(y_pred) - torch.log1p(y_true))**2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:12:05.343171Z","iopub.execute_input":"2024-12-03T11:12:05.343901Z","iopub.status.idle":"2024-12-03T11:12:05.350811Z","shell.execute_reply.started":"2024-12-03T11:12:05.343857Z","shell.execute_reply":"2024-12-03T11:12:05.349209Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Training function**","metadata":{}},{"cell_type":"code","source":"def train_model(model, train_loader, val_loader, epochs=50):\n    best_val_rmsle = float('inf')\n    for epoch in range(epochs):\n        model.train()\n        train_loss = 0.0\n\n        for X_batch, y_batch in train_loader:\n            X_batch, y_batch = X_batch.to(device), y_batch.to(device)\n\n            optimizer.zero_grad()\n            outputs = model(X_batch)\n            loss = criterion(outputs, y_batch)\n            loss.backward()\n            optimizer.step()\n\n            train_loss += loss.item()\n\n        model.eval()\n        val_loss = 0.0\n        val_rmsle = 0.0\n        with torch.no_grad():\n            for X_val_batch, y_val_batch in val_loader:\n                X_val_batch, y_val_batch = X_val_batch.to(device), y_val_batch.to(device)\n                val_outputs = model(X_val_batch)\n\n                val_loss += criterion(val_outputs, y_val_batch).item()\n                val_rmsle += calculate_rmsle(y_val_batch, val_outputs).item()\n\n        train_loss /= len(train_loader)\n        val_loss /= len(val_loader)\n        val_rmsle /= len(val_loader)\n\n        print(f\"Epoch {epoch + 1}/{epochs} - Train Loss: {train_loss:.4f} - Val Loss: {val_loss:.4f} - Val RMSLE: {val_rmsle:.4f}\")\n\n        if val_rmsle < best_val_rmsle:\n            best_val_rmsle = val_rmsle\n            torch.save(model.state_dict(), \"best_model.pth\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:12:09.871092Z","iopub.execute_input":"2024-12-03T11:12:09.871596Z","iopub.status.idle":"2024-12-03T11:12:09.881783Z","shell.execute_reply.started":"2024-12-03T11:12:09.871553Z","shell.execute_reply":"2024-12-03T11:12:09.880465Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"markdown","source":"You can use gpu for this, though it will not make much of a difference in terms of training time.\n\nAs my weekly quota got over within 3 days of use, I used cpu for training.","metadata":{}},{"cell_type":"markdown","source":"![](http://preview.redd.it/b7iagzpstbfa1.jpg?width=1080&crop=smart&auto=webp&s=6fbdab31605cfa3153211bf0065e6741029b8583)","metadata":{}},{"cell_type":"markdown","source":"Sadly, the above logic cannot work here🙃","metadata":{}},{"cell_type":"code","source":"train_model(model, train_loader, val_loader, epochs=25)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:12:31.691642Z","iopub.execute_input":"2024-12-03T11:12:31.692138Z","iopub.status.idle":"2024-12-03T11:36:56.706142Z","shell.execute_reply.started":"2024-12-03T11:12:31.692102Z","shell.execute_reply":"2024-12-03T11:36:56.704671Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Using the best model for further predictions on the test dataset**","metadata":{}},{"cell_type":"code","source":"model.load_state_dict(torch.load(\"best_model.pth\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:37:36.277567Z","iopub.execute_input":"2024-12-03T11:37:36.278769Z","iopub.status.idle":"2024-12-03T11:37:36.294892Z","shell.execute_reply.started":"2024-12-03T11:37:36.278681Z","shell.execute_reply":"2024-12-03T11:37:36.293588Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**The date-time column hadn't been encoded so I had to do it again**","metadata":{}},{"cell_type":"code","source":"for col in df_ts_encoded.columns:\n    if df_ts_encoded[col].dtype == 'object' or df_ts_encoded[col].dtype.name == 'category':\n        df_ts_encoded[col] = df_ts_encoded[col].astype('category').cat.codes\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:37:39.489505Z","iopub.execute_input":"2024-12-03T11:37:39.490041Z","iopub.status.idle":"2024-12-03T11:37:40.293057Z","shell.execute_reply.started":"2024-12-03T11:37:39.489997Z","shell.execute_reply":"2024-12-03T11:37:40.292086Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predicting on Test dataset","metadata":{}},{"cell_type":"code","source":"test_tensor = torch.tensor(df_ts_encoded.values, dtype=torch.float32).to(device)\ntest_predictions = torch.expm1(model(test_tensor)).cpu().detach().numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:37:43.867341Z","iopub.execute_input":"2024-12-03T11:37:43.867872Z","iopub.status.idle":"2024-12-03T11:37:50.649427Z","shell.execute_reply.started":"2024-12-03T11:37:43.867825Z","shell.execute_reply":"2024-12-03T11:37:50.648526Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# Prepare the submission file\ndf_s['Premium Amount'] = test_predictions\ndf_s.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T11:38:01.73268Z","iopub.execute_input":"2024-12-03T11:38:01.733172Z","iopub.status.idle":"2024-12-03T11:38:02.966205Z","shell.execute_reply.started":"2024-12-03T11:38:01.733135Z","shell.execute_reply":"2024-12-03T11:38:02.96491Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**If there are any suggestions you have for me, please write it down in the comments and I'll look into it.**\n\n**Do tell me ways I can improve this.**\n\n**Also, if you liked it, please upvote this notebook.Thank you!!**\n","metadata":{}},{"cell_type":"markdown","source":"![](http://i.imgflip.com/5ganuw.jpg)","metadata":{}}]}