{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":113002,"databundleVersionId":13471427,"isSourceIdPinned":false,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Title: \"Grand X-Ray Slam: Division B\"\n\n# Author: \"Ramandip Singh\"\n\n# Date: \"09-30-2025\"","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nfrom PIL import Image\nimport os\nimport timm\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T16:51:52.421325Z","iopub.execute_input":"2025-10-10T16:51:52.42182Z","iopub.status.idle":"2025-10-10T16:52:03.714979Z","shell.execute_reply.started":"2025-10-10T16:51:52.421796Z","shell.execute_reply":"2025-10-10T16:52:03.706245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 1. Configuration ---\n# IMPORTANT: Update these paths to match the Division B data\nDATA_DIR = \"/kaggle/input/grand-xray-slam-division-b/train2\"\nCSV_PATH = \"/kaggle/input/grand-xray-slam-division-b/train2.csv\"\n\nMODEL_NAME = \"efficientnet_b0\"\nIMAGE_SIZE = 256\nBATCH_SIZE = 32\nLEARNING_RATE = 1e-3\nEPOCHS = 10\nNUM_WORKERS = 2 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T16:52:03.716373Z","iopub.execute_input":"2025-10-10T16:52:03.716863Z","iopub.status.idle":"2025-10-10T16:52:03.721325Z","shell.execute_reply.started":"2025-10-10T16:52:03.716843Z","shell.execute_reply":"2025-10-10T16:52:03.720456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 2. Load Data and Define Labels ---\ndf = pd.read_csv(CSV_PATH)\n\n# IMPORTANT: This line helps you find the correct column names to prevent errors!\nprint(\"Columns in your CSV file:\")\nprint(df.columns)\nprint(\"-\" * 25)\n\nLABELS = [\n    'Atelectasis', 'Cardiomegaly', 'Consolidation', 'Edema',\n    'Enlarged Cardiomediastinum', 'Fracture', 'Lung Lesion',\n    'Lung Opacity', 'Pleural Effusion', 'Pleural Other',\n    'Pneumonia', 'Pneumothorax', 'Support Devices', 'No Finding'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T16:52:03.72215Z","iopub.execute_input":"2025-10-10T16:52:03.722946Z","iopub.status.idle":"2025-10-10T16:52:04.043068Z","shell.execute_reply.started":"2025-10-10T16:52:03.722924Z","shell.execute_reply":"2025-10-10T16:52:04.042399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ChestXRayDataset(Dataset):\n    def __init__(self, dataframe, image_dir, labels, transform=None):\n        self.df = dataframe\n        self.image_dir = image_dir\n        self.transform = transform\n        self.labels = labels\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        correct_column_name = 'Image_name' \n        img_path = os.path.join(self.image_dir, self.df.iloc[idx][correct_column_name])\n\n        try:\n            # --- Try to open the image ---\n            image = Image.open(img_path).convert(\"RGB\")\n        except Exception as e:\n            # --- If it fails, handle the error ---\n            print(f\"Warning: Could not load image {img_path}. Error: {e}. Loading a replacement.\")\n            # Load the first image of the dataset as a fallback\n            replacement_path = os.path.join(self.image_dir, self.df.iloc[0][correct_column_name])\n            image = Image.open(replacement_path).convert(\"RGB\")\n\n        # Get the labels for the ORIGINAL index\n        labels_vector = self.df.iloc[idx][self.labels].values.astype(np.float32)\n        labels_tensor = torch.tensor(labels_vector, dtype=torch.float32)\n\n        if self.transform:\n            image = self.transform(image)\n\n        return image, labels_tensor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T16:52:04.044577Z","iopub.execute_input":"2025-10-10T16:52:04.044801Z","iopub.status.idle":"2025-10-10T16:52:04.05119Z","shell.execute_reply.started":"2025-10-10T16:52:04.044784Z","shell.execute_reply":"2025-10-10T16:52:04.050408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# --- 4. Transforms and Data Augmentation ---\nmean = [0.485, 0.456, 0.406]\nstd = [0.229, 0.224, 0.225]\n\ntrain_transform = transforms.Compose([\n    transforms.Resize((IMAGE_SIZE, IMAGE_SIZE)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(10),\n    transforms.ToTensor(),\n    transforms.Normalize(mean, std)\n])\n\nval_transform = transforms.Compose([\n    transforms.Resize((IMAGE_SIZE, IMAGE_SIZE)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean, std)\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T16:52:04.051863Z","iopub.execute_input":"2025-10-10T16:52:04.052081Z","iopub.status.idle":"2025-10-10T16:52:04.070667Z","shell.execute_reply.started":"2025-10-10T16:52:04.052064Z","shell.execute_reply":"2025-10-10T16:52:04.070028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 5. Data Splitting and Loaders ---\ntrain_df, val_df = train_test_split(df, test_size=0.2, random_state=42)\n\ntrain_dataset = ChestXRayDataset(train_df, DATA_DIR, LABELS, transform=train_transform)\nval_dataset = ChestXRayDataset(val_df, DATA_DIR, LABELS, transform=val_transform)\n\ntrain_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True, num_workers=NUM_WORKERS)\nval_loader = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False, num_workers=NUM_WORKERS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T16:52:04.0714Z","iopub.execute_input":"2025-10-10T16:52:04.071627Z","iopub.status.idle":"2025-10-10T16:52:04.114335Z","shell.execute_reply.started":"2025-10-10T16:52:04.071611Z","shell.execute_reply":"2025-10-10T16:52:04.113766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 6. Model, Loss Function, and Optimizer ---\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nmodel = timm.create_model(MODEL_NAME, pretrained=True, num_classes=len(LABELS))\nmodel.to(device)\n\n# Calculate positive weights to handle class imbalance\npos_weights = []\nfor label in LABELS:\n    num_pos = df[label].sum()\n    num_neg = len(df) - num_pos\n    pos_weights.append(num_neg / (num_pos + 1e-6)) # Added epsilon for stability\npos_weights = torch.tensor(pos_weights, dtype=torch.float32).to(device)\n\ncriterion = nn.BCEWithLogitsLoss(pos_weight=pos_weights)\noptimizer = torch.optim.AdamW(model.parameters(), lr=LEARNING_RATE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T16:52:04.115412Z","iopub.execute_input":"2025-10-10T16:52:04.115667Z","iopub.status.idle":"2025-10-10T16:52:06.469606Z","shell.execute_reply.started":"2025-10-10T16:52:04.115638Z","shell.execute_reply":"2025-10-10T16:52:06.469022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 7. Training and Validation Functions ---\ndef train_one_epoch(model, loader, optimizer, criterion, device):\n    model.train()\n    running_loss = 0.0\n    for images, labels in loader:\n        images, labels = images.to(device), labels.to(device)\n        optimizer.zero_grad()\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item()\n    return running_loss / len(loader)\n\ndef validate_one_epoch(model, loader, criterion, device):\n    model.eval()\n    running_loss = 0.0\n    with torch.no_grad():\n        for images, labels in loader:\n            images, labels = images.to(device), labels.to(device)\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            running_loss += loss.item()\n    return running_loss / len(loader)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T16:52:06.470598Z","iopub.execute_input":"2025-10-10T16:52:06.470814Z","iopub.status.idle":"2025-10-10T16:52:06.476204Z","shell.execute_reply.started":"2025-10-10T16:52:06.470798Z","shell.execute_reply":"2025-10-10T16:52:06.475223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 8. Main Training Loop ---\nbest_val_loss = float('inf')\n\nfor epoch in range(EPOCHS):\n    print(f\"--- Epoch {epoch+1}/{EPOCHS} ---\")\n    \n    train_loss = train_one_epoch(model, train_loader, optimizer, criterion, device)\n    print(f\"Epoch {epoch+1} Training Loss: {train_loss:.4f}\")\n\n    val_loss = validate_one_epoch(model, val_loader, criterion, device)\n    print(f\"Epoch {epoch+1} Validation Loss: {val_loss:.4f}\")\n\n    if val_loss < best_val_loss:\n        best_val_loss = val_loss\n        torch.save(model.state_dict(), \"best_model.pth\")\n        print(f\"New best model saved with validation loss: {best_val_loss:.4f}\")\n\nprint(\"\\nFinished Training\")\nprint(f\"Best validation loss achieved: {best_val_loss:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T16:52:06.477052Z","iopub.execute_input":"2025-10-10T16:52:06.477314Z","iopub.status.idle":"2025-10-11T03:20:39.372172Z","shell.execute_reply.started":"2025-10-10T16:52:06.477291Z","shell.execute_reply":"2025-10-11T03:20:39.371114Z"}},"outputs":[],"execution_count":null}]}