{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":112899,"databundleVersionId":13449579,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nimport torchvision\nimport torch.nn as nn\nimport os\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:16.554076Z","iopub.execute_input":"2025-10-07T09:44:16.554636Z","iopub.status.idle":"2025-10-07T09:44:16.559153Z","shell.execute_reply.started":"2025-10-07T09:44:16.554612Z","shell.execute_reply":"2025-10-07T09:44:16.558618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndf = pd.read_csv('/kaggle/input/grand-xray-slam-division-a/train1.csv')\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:16.665513Z","iopub.execute_input":"2025-10-07T09:44:16.665809Z","iopub.status.idle":"2025-10-07T09:44:16.880974Z","shell.execute_reply.started":"2025-10-07T09:44:16.665789Z","shell.execute_reply":"2025-10-07T09:44:16.880326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:16.88211Z","iopub.execute_input":"2025-10-07T09:44:16.882406Z","iopub.status.idle":"2025-10-07T09:44:16.886805Z","shell.execute_reply.started":"2025-10-07T09:44:16.882383Z","shell.execute_reply":"2025-10-07T09:44:16.886254Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Setting up config","metadata":{}},{"cell_type":"code","source":"###### Configuration setup\nBASE_PATH = '/kaggle/input/grand-xray-slam-division-a/'\nTRAIN_IMG_PATH = os.path.join(BASE_PATH, 'train1/')\nTEST_IMG_PATH = os.path.join(BASE_PATH, 'test1/')\nLABELS = [\n    'Atelectasis', 'Cardiomegaly', 'Consolidation', 'Edema', 'Enlarged Cardiomediastinum', 'Fracture',\n    'Lung Lesion', 'Lung Opacity', 'No Finding', 'Pleural Effusion', 'Pleural Other', 'Pneumonia', 'Pneumothorax', 'Support Devices'\n]\nDEVICE = 'cuda' if torch.cuda.is_available() else 'cpu'\nNUM_WORKERS = int(os.cpu_count() // 2)\nLR = 0.001\nWEIGHT_DECAY = 0.0001\nEPOCHS = 4\nBATCH_SIZE = 16","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:17.039851Z","iopub.execute_input":"2025-10-07T09:44:17.040104Z","iopub.status.idle":"2025-10-07T09:44:17.044894Z","shell.execute_reply.started":"2025-10-07T09:44:17.040084Z","shell.execute_reply":"2025-10-07T09:44:17.044305Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Setting up Dataset/Dataloaders","metadata":{}},{"cell_type":"code","source":"class XRayDataset(Dataset):\n    def __init__(self, df, dr = TRAIN_IMG_PATH, is_Test = False, transforms = None):\n        self.df = df\n        self.is_Test = is_Test\n        self.labels = LABELS\n        self.dr = dr\n        self.transforms = transforms\n\n    def __len__(self):\n        return len(self.df)\n\n    def load_image(self, idx):\n        img_path = self.df.iloc[idx]['Image_name']\n        img = Image.open(img_path).convert('RGB')\n        return img\n        \n    def __getitem__(self, idx):\n        img_path = self.df.iloc[idx]['Image_name']\n        img_path = os.path.join(self.dr, img_path)\n        img = Image.open(img_path).convert('RGB')\n        label = self.df.iloc[idx][LABELS].values.astype(np.float32)\n        if self.transforms:\n            img =  self.transforms(img)\n        if not self.is_Test:\n            return img, torch.tensor(label, dtype = torch.float32)\n        else:\n            return img\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T14:30:00.977482Z","iopub.execute_input":"2025-10-07T14:30:00.977785Z","iopub.status.idle":"2025-10-07T14:30:00.984474Z","shell.execute_reply.started":"2025-10-07T14:30:00.977761Z","shell.execute_reply":"2025-10-07T14:30:00.983692Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2.1 Displaying label and image","metadata":{}},{"cell_type":"code","source":"transform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor()\n])\nobj = XRayDataset(df[:5], is_Test=False, transforms=transform)\nimg, label = obj[2]\nclasses = np.where(np.array(label) == 1)[0]\nlabel_names = [LABELS[i] for i in classes]\nprint(classes)\nprint(label_names)\nplt.imshow(img.permute(1,2, 0))\nplt.axis('off')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:17.5555Z","iopub.execute_input":"2025-10-07T09:44:17.555773Z","iopub.status.idle":"2025-10-07T09:44:17.695601Z","shell.execute_reply.started":"2025-10-07T09:44:17.555754Z","shell.execute_reply":"2025-10-07T09:44:17.694865Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Getting the model","metadata":{}},{"cell_type":"markdown","source":"## 3.1 Model = efficientnet_v2_s\n[https://docs.pytorch.org/vision/main/models/generated/torchvision.models.efficientnet_v2_s.html#efficientnet-v2-s](http://)","metadata":{}},{"cell_type":"code","source":"weights_effnet = torchvision.models.EfficientNet_V2_S_Weights.DEFAULT\neffnet_model = torchvision.models.efficientnet_v2_s(weights = weights_effnet)\n# effnet_model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:17.923695Z","iopub.execute_input":"2025-10-07T09:44:17.923911Z","iopub.status.idle":"2025-10-07T09:44:18.386428Z","shell.execute_reply.started":"2025-10-07T09:44:17.923897Z","shell.execute_reply":"2025-10-07T09:44:18.385711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"effnet_transforms = weights_effnet.transforms()\neffnet_transforms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:18.387783Z","iopub.execute_input":"2025-10-07T09:44:18.388055Z","iopub.status.idle":"2025-10-07T09:44:18.392914Z","shell.execute_reply.started":"2025-10-07T09:44:18.388032Z","shell.execute_reply":"2025-10-07T09:44:18.392092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for param in effnet_model.features.parameters():\n    param.requires_grad = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:18.393589Z","iopub.execute_input":"2025-10-07T09:44:18.393764Z","iopub.status.idle":"2025-10-07T09:44:18.408614Z","shell.execute_reply.started":"2025-10-07T09:44:18.393749Z","shell.execute_reply":"2025-10-07T09:44:18.407882Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3.2 Model Summary","metadata":{}},{"cell_type":"code","source":"## get model summary\ntry:\n    from torchinfo import summary\nexcept:\n    print(\"[INFO] Couldn't find torchinfo... installing it.\")\n    !pip install -q torchinfo\n    from torchinfo import summary\nsummary(model=effnet_model, \n        input_size=(32, 3, 224, 224), # make sure this is \"input_size\", not \"input_shape\"\n        # col_names=[\"input_size\"], # uncomment for smaller output\n        col_names=[\"input_size\", \"output_size\", \"num_params\", \"trainable\"],\n        col_width=10,\n        row_settings=[\"var_names\"]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:18.464652Z","iopub.execute_input":"2025-10-07T09:44:18.464852Z","iopub.status.idle":"2025-10-07T09:44:18.668647Z","shell.execute_reply.started":"2025-10-07T09:44:18.464837Z","shell.execute_reply":"2025-10-07T09:44:18.667907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"effnet_model.classifier = nn.Sequential(\n    torch.nn.Dropout(p=0.2, inplace=True),\n    torch.nn.Linear(in_features = 1280,\n                   out_features = len(LABELS),\n                   bias = True)\n).to(DEVICE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:18.669814Z","iopub.execute_input":"2025-10-07T09:44:18.670028Z","iopub.status.idle":"2025-10-07T09:44:18.674577Z","shell.execute_reply.started":"2025-10-07T09:44:18.670012Z","shell.execute_reply":"2025-10-07T09:44:18.673759Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3.3 Optimzer and Loss","metadata":{}},{"cell_type":"code","source":"loss_fn = nn.BCEWithLogitsLoss()\noptimizer = torch.optim.Adam(effnet_model.parameters(), lr = LR, weight_decay = WEIGHT_DECAY)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:18.870102Z","iopub.execute_input":"2025-10-07T09:44:18.87055Z","iopub.status.idle":"2025-10-07T09:44:18.875787Z","shell.execute_reply.started":"2025-10-07T09:44:18.87053Z","shell.execute_reply":"2025-10-07T09:44:18.875072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:19.002201Z","iopub.execute_input":"2025-10-07T09:44:19.00288Z","iopub.status.idle":"2025-10-07T09:44:19.007287Z","shell.execute_reply.started":"2025-10-07T09:44:19.002852Z","shell.execute_reply":"2025-10-07T09:44:19.006643Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Setting up training and testing","metadata":{}},{"cell_type":"code","source":"def train_step(model: torch.nn.Module, \n              dataloader : torch.utils.data.DataLoader,\n              loss_fn : torch.nn.Module, \n              optimizer : torch.optim.Optimizer, \n              device : torch.device):\n    \n    model.train()\n    train_loss, train_acc = 0, 0\n\n    for batch, (X, y) in enumerate(dataloader):\n        X, y, = X.to(device), y.to(device)\n\n        ## fwd pass\n        y_pred = model(X)\n\n        loss = loss_fn(y_pred, y)\n        train_loss += loss.item()\n\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        ## convert logits to probs\n        y_pred_probs = torch.sigmoid(y_pred)\n        y_pred_class = (y_pred_probs > 0.5).float()\n\n        correct = (y_pred_class == y).float().mean()\n        train_acc += correct.item()\n        # print(f\"Train: {batch}\")\n        \n    train_loss = train_loss / len(dataloader)\n    train_acc = train_acc / len(dataloader)\n\n    return train_loss, train_acc\n    \ndef test_step(model: torch.nn.Module, \n             dataloader: torch.utils.data.DataLoader, \n             loss_fn: torch.nn.Module, \n             device : torch.device):\n\n    model.eval()\n\n    test_loss, test_acc = 0, 0\n    \n    with torch.inference_mode():\n        for batch, (X, y) in enumerate(dataloader):\n            X, y = X.to(device), y.to(device)\n\n            test_pred_logits = model(X)\n\n            loss = loss_fn(test_pred_logits, y)\n            test_loss += loss.item()\n\n            test_pred_labels = torch.sigmoid(test_pred_logits)\n            test_pred_class = (test_pred_labels > 0.5).float()\n\n            test_acc += (test_pred_class == y).float().mean().item()\n            # print(f\"Test: {batch}\")\n\n    test_loss = test_loss / len(dataloader)\n    test_acc = test_acc / len(dataloader)\n    \n    return test_loss, test_acc\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:19.292005Z","iopub.execute_input":"2025-10-07T09:44:19.292241Z","iopub.status.idle":"2025-10-07T09:44:19.299851Z","shell.execute_reply.started":"2025-10-07T09:44:19.292218Z","shell.execute_reply":"2025-10-07T09:44:19.299094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train(model, train_dataloader, test_dataloader, optimizer, loss_fn, epochs, device):\n    results = {\n        'train_loss' : [],\n        'train_acc' : [],\n        'test_loss' : [],\n        'test_acc': []\n    }\n\n    ## to save model \n    best_loss = float('inf')\n    model_name = model.__class__.__name__\n    best_model_path = f\"{model_name}_best.pth\"\n    checkpoint_path = f\"{model_name}_checkpoint.pth\"\n\n    for epoch in tqdm(range(epochs)):\n        train_loss, train_acc = train_step(model=model, \n                                          dataloader=train_dataloader,\n                                          loss_fn = loss_fn,\n                                          optimizer=optimizer,\n                                          device=device)\n        test_loss, test_acc = test_step(model=model, \n                                       dataloader = test_dataloader,\n                                       loss_fn = loss_fn,\n                                       device=device)\n\n        ## printing results\n        print(f\"Epoch: {epoch+1} | train_loss: {train_loss:.4f} | test_loss: {test_loss:.4f} | test_acc: {test_acc:.4f}\")\n\n        results[\"train_loss\"].append(train_loss)\n        results[\"train_acc\"].append(train_acc)\n        results[\"test_loss\"].append(test_loss)\n        results[\"test_acc\"].append(test_acc)\n\n        if test_loss < best_loss:\n            best_loss = test_loss\n            torch.save(model.state_dict(), best_model_path)\n            print(f\"Save best model: {best_model_path} (epoch {epoch + 1})\")\n        \n        ## checkpoint to resume\n        checkpoint = {\n            'epoch' : epoch,\n            'model_state_dict' : model.state_dict(),\n            'optimizer_state_dict' : optimizer.state_dict(),\n            'best_loss' : best_loss\n        }\n        torch.save(checkpoint, checkpoint_path)\n\n    print(f\"Training Complete.... Best Model saved as: {best_model_path}\")\n    return results","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:19.414546Z","iopub.execute_input":"2025-10-07T09:44:19.414746Z","iopub.status.idle":"2025-10-07T09:44:19.421348Z","shell.execute_reply.started":"2025-10-07T09:44:19.414731Z","shell.execute_reply":"2025-10-07T09:44:19.420646Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4.1 setting up dataloaders and training","metadata":{}},{"cell_type":"code","source":"train_df, test_df = train_test_split(df, test_size=0.1, shuffle=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:27.792621Z","iopub.execute_input":"2025-10-07T09:44:27.793126Z","iopub.status.idle":"2025-10-07T09:44:27.818278Z","shell.execute_reply.started":"2025-10-07T09:44:27.793102Z","shell.execute_reply":"2025-10-07T09:44:27.817658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = XRayDataset(train_df, transforms=effnet_transforms)\ntest_data = XRayDataset(test_df, transforms=effnet_transforms)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:27.935073Z","iopub.execute_input":"2025-10-07T09:44:27.935324Z","iopub.status.idle":"2025-10-07T09:44:27.938913Z","shell.execute_reply.started":"2025-10-07T09:44:27.935307Z","shell.execute_reply":"2025-10-07T09:44:27.938335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataloader = DataLoader(train_data, batch_size=BATCH_SIZE, num_workers=NUM_WORKERS, pin_memory=True)\ntest_dataloader = DataLoader(test_data, batch_size=BATCH_SIZE, num_workers=NUM_WORKERS, pin_memory=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:28.23061Z","iopub.execute_input":"2025-10-07T09:44:28.23088Z","iopub.status.idle":"2025-10-07T09:44:28.234984Z","shell.execute_reply.started":"2025-10-07T09:44:28.230861Z","shell.execute_reply":"2025-10-07T09:44:28.234302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_dataloader), len(test_dataloader)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:28.639417Z","iopub.execute_input":"2025-10-07T09:44:28.639686Z","iopub.status.idle":"2025-10-07T09:44:28.644589Z","shell.execute_reply.started":"2025-10-07T09:44:28.639666Z","shell.execute_reply":"2025-10-07T09:44:28.643753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = train(effnet_model, train_dataloader, test_dataloader, optimizer=optimizer, loss_fn=loss_fn, epochs = EPOCHS, device=DEVICE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T09:44:29.545468Z","iopub.execute_input":"2025-10-07T09:44:29.54574Z","iopub.status.idle":"2025-10-07T14:21:15.987353Z","shell.execute_reply.started":"2025-10-07T09:44:29.545721Z","shell.execute_reply":"2025-10-07T14:21:15.986254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(12, 8))\n\naxes[0].plot(results['train_loss'], color='blue')\naxes[0].set_title('Train Loss')\naxes[0].set_xlabel('Epochs')\naxes[0].set_ylabel('Train Loss')\n\naxes[1].plot(results['test_loss'], color='green')\naxes[1].set_title('Test Loss')\naxes[1].set_xlabel('Epochs')\naxes[1].set_ylabel('Test Loss')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T15:18:40.967367Z","iopub.execute_input":"2025-10-07T15:18:40.967648Z","iopub.status.idle":"2025-10-07T15:18:41.310209Z","shell.execute_reply.started":"2025-10-07T15:18:40.967627Z","shell.execute_reply":"2025-10-07T15:18:41.309531Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Submission pipeline","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/grand-xray-slam-division-a/sample_submission_1.csv')\nfinal_dataset = XRayDataset(sample_submission, dr=TEST_IMG_PATH, is_Test = True, transforms = effnet_transforms)\nfinal_dataloader = DataLoader(final_dataset, batch_size=BATCH_SIZE, num_workers=NUM_WORKERS, shuffle=False)\neffnet_model.eval()\npredictions=[]\n\nwith torch.inference_mode():\n    for batch, X in enumerate(final_dataloader):\n        X = X.to(DEVICE)\n        pred_probs = effnet_model(X)\n        preds = torch.sigmoid(pred_probs)\n        predictions.append(preds.cpu().numpy())\n\npredictions = np.concatenate(predictions, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T14:30:41.774376Z","iopub.execute_input":"2025-10-07T14:30:41.774982Z","iopub.status.idle":"2025-10-07T15:02:31.561828Z","shell.execute_reply.started":"2025-10-07T14:30:41.774952Z","shell.execute_reply":"2025-10-07T15:02:31.560905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission = sample_submission.copy()\nfinal_submission[LABELS] = predictions\nfinal_submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T15:02:31.564032Z","iopub.execute_input":"2025-10-07T15:02:31.564302Z","iopub.status.idle":"2025-10-07T15:02:31.591585Z","shell.execute_reply.started":"2025-10-07T15:02:31.564276Z","shell.execute_reply":"2025-10-07T15:02:31.590981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission.to_csv('submission.csv', index=False)\nprint('Submission csv created... THE END !!')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T15:05:16.714703Z","iopub.execute_input":"2025-10-07T15:05:16.715026Z","iopub.status.idle":"2025-10-07T15:05:17.415003Z","shell.execute_reply.started":"2025-10-07T15:05:16.715002Z","shell.execute_reply":"2025-10-07T15:05:17.413983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}