{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":12264896,"datasetId":7588675,"databundleVersionId":12813681},{"sourceType":"datasetVersion","sourceId":1810938,"datasetId":1075803,"databundleVersionId":1848422},{"sourceType":"datasetVersion","sourceId":14585127,"datasetId":9316745,"databundleVersionId":15419272},{"sourceType":"modelInstanceVersion","sourceId":438484,"databundleVersionId":12739238,"modelInstanceId":357737}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\nfrom PIL import Image\nfrom tqdm.notebook import tqdm\nimport pydicom\nimport torch\nfrom PIL import Image\nimport os\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom PIL import Image\nimport os\nimport torch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:21.383586Z","iopub.execute_input":"2026-01-24T14:03:21.384177Z","iopub.status.idle":"2026-01-24T14:03:27.069655Z","shell.execute_reply.started":"2026-01-24T14:03:21.384152Z","shell.execute_reply":"2026-01-24T14:03:27.069029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# Base directory (adjust if necessary)\nPNG_FOLDER = \"/kaggle/input/vinbigdata-chest-xray-resized-png-1024x1024/train\"\n\n\nBASE_DIR_csv = \"/kaggle/input/csv-data\"\nTRAIN_CSV = os.path.join(BASE_DIR_csv, \"train.csv\")\n\n# Load CSVs\ntrain_df = pd.read_csv(TRAIN_CSV)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:27.07079Z","iopub.execute_input":"2026-01-24T14:03:27.071196Z","iopub.status.idle":"2026-01-24T14:03:27.214197Z","shell.execute_reply.started":"2026-01-24T14:03:27.071166Z","shell.execute_reply":"2026-01-24T14:03:27.213635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unique_scans = train_df[\"image_id\"].nunique()\nprint(f\"Number of unique chest X-ray scans in train.csv: {unique_scans}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:27.214826Z","iopub.execute_input":"2026-01-24T14:03:27.215031Z","iopub.status.idle":"2026-01-24T14:03:27.232417Z","shell.execute_reply.started":"2026-01-24T14:03:27.215015Z","shell.execute_reply":"2026-01-24T14:03:27.231733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by class_name and count unique image_ids\nunique_per_class = train_df.groupby(\"class_name\")[\"image_id\"].nunique().sort_values(ascending=False)\n\n# Convert to DataFrame for display\nunique_per_class_df = unique_per_class.reset_index()\nunique_per_class_df.columns = [\"Class Name\", \"Unique Scan Count\"]\n\n# Display\nprint(unique_per_class_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:27.234054Z","iopub.execute_input":"2026-01-24T14:03:27.234301Z","iopub.status.idle":"2026-01-24T14:03:27.262118Z","shell.execute_reply.started":"2026-01-24T14:03:27.234283Z","shell.execute_reply":"2026-01-24T14:03:27.261579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ntrain_df = pd.read_csv(\"/kaggle/input/vinbigdata-splitting-diseases/train_balanced_multilabel.csv\")\ntest_df  = pd.read_csv(\"/kaggle/input/vinbigdata-splitting-diseases/test_balanced_multilabel.csv\")\n\ndef add_binary_label(df):\n    df[\"binary_label\"] = (df[\"No finding\"] == 0).astype(int)\n    return df\n\ntrain_df = add_binary_label(train_df)\ntest_df  = add_binary_label(test_df)\n\ntrain_df = train_df[[\"image_id\", \"binary_label\"]]\ntest_df  = test_df[[\"image_id\", \"binary_label\"]]\n\nprint(\"Train distribution:\")\nprint(train_df[\"binary_label\"].value_counts())\n\nprint(\"\\nTest distribution:\")\nprint(test_df[\"binary_label\"].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:27.262671Z","iopub.execute_input":"2026-01-24T14:03:27.262936Z","iopub.status.idle":"2026-01-24T14:03:27.313288Z","shell.execute_reply.started":"2026-01-24T14:03:27.262919Z","shell.execute_reply":"2026-01-24T14:03:27.312746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:27.313856Z","iopub.execute_input":"2026-01-24T14:03:27.314015Z","iopub.status.idle":"2026-01-24T14:03:27.327392Z","shell.execute_reply.started":"2026-01-24T14:03:27.314001Z","shell.execute_reply":"2026-01-24T14:03:27.326719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# from PIL import Image\n# import numpy as np\n# from tqdm import tqdm\n\n# IMG_DIR = \"/kaggle/input/vinbigdata-chest-xray-resized-png-1024x1024/train\"  # change if needed\n\n# image_ids = train_df[\"image_id\"].values\n\n# sums = np.zeros(3)\n# sqs = np.zeros(3)\n# n_images = len(image_ids)\n\n# for img_id in tqdm(image_ids, desc=\"Calculating mean/std\"):\n    \n#     path = os.path.join(IMG_DIR, img_id + \".png\")  # build path\n    \n#     img = Image.open(path).convert(\"RGB\").resize((512, 512))\n#     img = np.array(img, dtype=np.float32) / 255.0\n    \n#     sums += img.mean(axis=(0, 1))\n#     sqs += (img ** 2).mean(axis=(0, 1))\n\n# mean = sums / n_images\n# std = np.sqrt(sqs / n_images - mean ** 2)\n\n# print(\"Mean:\", mean)\n# print(\"Std :\", std)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:27.328235Z","iopub.execute_input":"2026-01-24T14:03:27.328455Z","iopub.status.idle":"2026-01-24T14:03:27.332151Z","shell.execute_reply.started":"2026-01-24T14:03:27.328432Z","shell.execute_reply":"2026-01-24T14:03:27.331555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# BATCH_SIZE = 32\nNUM_WORKERS = 1\nEPOCHS = 35          # For demo; increase for real runs\nLR = 5e-5\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nSEED = 21\nIMG_SIZE = 512      # ResNet input size\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:27.332872Z","iopub.execute_input":"2026-01-24T14:03:27.333106Z","iopub.status.idle":"2026-01-24T14:03:27.426717Z","shell.execute_reply.started":"2026-01-24T14:03:27.333089Z","shell.execute_reply":"2026-01-24T14:03:27.426005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nfrom PIL import Image\n\nmy_mean= [0.54881824 ,0.54881824  ,0.54881824 ]\nmy_std= [0.26736849 ,0.26736849,0.26736849]\n\nimport os\nfrom torch.utils.data import Dataset\nfrom PIL import Image\n\nclass VinBigDataset(Dataset):\n    def __init__(self, df, img_dir, transform=None):\n        self.df = df.reset_index(drop=True)\n        self.img_dir = img_dir          # 👈 NEW\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        image_id = self.df.loc[idx, \"image_id\"]\n        label = self.df.loc[idx, \"binary_label\"]\n\n        path = os.path.join(self.img_dir, image_id + \".png\")  # 👈 build path\n\n        img = Image.open(path).convert(\"RGB\")\n\n        if self.transform:\n            img = self.transform(img)\n\n        return img, label\n\n# Transforms\ntrain_transform = transforms.Compose([\n    \n    transforms.RandomResizedCrop(IMG_SIZE, scale=(0.8, 1.0)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(degrees=15),\n    transforms.ColorJitter(brightness=0.2, contrast=0.2, saturation=0.2),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=my_mean, std=my_std),\n    transforms.Resize((IMG_SIZE, IMG_SIZE))\n])\nval_transform = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=my_mean, std=my_std)\n])\n\n\n\n\n\nimg_dir=\"/kaggle/input/vinbigdata-chest-xray-resized-png-1024x1024/train\"\ntrain_set = VinBigDataset(train_df, img_dir, transform=train_transform)\nval_set   = VinBigDataset(test_df,   img_dir, transform=val_transform)\n\ntrain_loader = DataLoader(train_set, batch_size=2, shuffle=True, num_workers=NUM_WORKERS)\nval_loader   = DataLoader(val_set, batch_size=2, shuffle=False, num_workers=NUM_WORKERS)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:27.427439Z","iopub.execute_input":"2026-01-24T14:03:27.427758Z","iopub.status.idle":"2026-01-24T14:03:30.577326Z","shell.execute_reply.started":"2026-01-24T14:03:27.427735Z","shell.execute_reply":"2026-01-24T14:03:30.576531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torchvision import models\nimport timm\nfrom torch import nn\nfrom torchvision import models\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau  # Import scheduler\nfrom torch.optim.lr_scheduler import CosineAnnealingWarmRestarts\n# =============== MODEL ===============\nfrom torch import nn\nfrom torchvision import models\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau  # Import scheduler\n\n\n\n\n\nmodel = models.resnet101(weights=\"IMAGENET1K_V1\")\nin_features = model.fc.in_features  # Correct: resnet34 uses .fc, not .classifier\n\n    # Replace the final fully connected layer\nmodel.fc = nn.Sequential(\n        nn.Dropout(0.5),            # Dropout for regularization\n        nn.Linear(in_features, 2)   # 2 output classes\n    )\n    \n\n# =============== LOSS, OPTIM ===============\ncriterion = torch.nn.CrossEntropyLoss()\noptimizer = torch.optim.AdamW(model.parameters(), lr=LR, weight_decay=1e-4)\n\nif torch.cuda.device_count() > 1:\n    print(f\"Using {torch.cuda.device_count()} GPUs\")\n    model = nn.DataParallel(model)\nmodel = model.to(DEVICE)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:30.580144Z","iopub.execute_input":"2026-01-24T14:03:30.580537Z","iopub.status.idle":"2026-01-24T14:03:35.801004Z","shell.execute_reply.started":"2026-01-24T14:03:30.580519Z","shell.execute_reply":"2026-01-24T14:03:35.800258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import DataLoader\nfrom sklearn.metrics import accuracy_score, f1_score\nfrom tqdm import tqdm\n\n# =========================\n# ======== SETTINGS =======\n# =========================\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nBATCH_SIZE = 2                 # small batch size for GPU\nEFFECTIVE_BATCH_SIZE = 32      # update weights after 32 images\naccumulation_steps = EFFECTIVE_BATCH_SIZE // BATCH_SIZE\n\nEPOCHS = 35\nearly_stop_patience = 10\n\n# =========================\n# ===== DATA LOADERS =====\n# =========================\n# train_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True, num_workers=2, pin_memory=True)\n# val_loader = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False, num_workers=2, pin_memory=True)\n\nprint(f\"Accumulating gradients for {accumulation_steps} mini-batches to simulate batch size {EFFECTIVE_BATCH_SIZE}\")\n# Freeze BatchNorm helper\ndef set_bn_eval(m):\n    if isinstance(m, torch.nn.modules.batchnorm._BatchNorm):\n        m.eval()\n\n# =========================\n# ===== FNS: TRAIN/EVAL ===\n# =========================\ndef train_one_epoch(model, loader, optimizer, criterion, accumulation_steps):\n    model.train()\n    model.apply(set_bn_eval) \n    # [CRITICAL FIX] Freeze Batch Norm layers because Batch Size 2 is too small\n    # This prevents the model from learning \"noise\" from the small batches\n\n    model.apply(set_bn_eval) \n\n    running_loss = 0\n    all_preds, all_labels = [], []\n\n    optimizer.zero_grad()\n    \n    for i, (imgs, labels) in enumerate(tqdm(loader, desc=\"Training\")):\n        imgs, labels = imgs.to(DEVICE), labels.to(DEVICE)\n        \n        # 1. Forward Pass\n        outputs = model(imgs)\n        \n        # 2. Scale Loss\n        loss = criterion(outputs, labels) / accumulation_steps \n        \n        # 3. Backward Pass\n        loss.backward()\n\n        # Track Loss (revert scaling for reporting)\n        running_loss += loss.item() * accumulation_steps * imgs.size(0)\n        \n        # Track Metrics\n        preds = outputs.argmax(1).detach().cpu().numpy()\n        all_preds.extend(preds)\n        all_labels.extend(labels.cpu().numpy())\n\n        # 4. Step and Zero Grad\n        # Check if it's an accumulation step OR the very last batch\n        if (i + 1) % accumulation_steps == 0 or (i + 1) == len(loader):\n            optimizer.step()\n            optimizer.zero_grad()\n\n    epoch_loss = running_loss / len(loader.dataset)\n    acc = accuracy_score(all_labels, all_preds)\n    f1 = f1_score(all_labels, all_preds, average='weighted')\n    \n    return epoch_loss, acc, f1\n\ndef eval_one_epoch(model, loader, criterion):\n    model.eval()\n    running_loss = 0\n    all_preds, all_labels = [], []\n    with torch.no_grad():\n        for imgs, labels in tqdm(loader, desc=\"Evaluating\"):\n            imgs, labels = imgs.to(DEVICE), labels.to(DEVICE)\n            outputs = model(imgs)\n            loss = criterion(outputs, labels)\n            running_loss += loss.item() * imgs.size(0)\n            preds = outputs.argmax(1).detach().cpu().numpy()\n            all_preds.extend(preds)\n            all_labels.extend(labels.cpu().numpy())\n    epoch_loss = running_loss / len(loader.dataset)\n    acc = accuracy_score(all_labels, all_preds)\n    f1 = f1_score(all_labels, all_preds, average='weighted')\n    return epoch_loss, acc, f1, all_preds, all_labels\n\n# =========================\n# ===== TRAINING LOOP =====\n# =========================\ntrain_losses, val_losses = [], []\ntrain_accs, val_accs = [], []\ntrain_f1s, val_f1s = [], []\n\nbest_val_f1 = 0.0\nepochs_no_improve = 0\n\nfor epoch in range(EPOCHS):\n    print(f\"\\nEpoch {epoch+1}/{EPOCHS}\")\n\n    # === TRAINING\n    tr_loss, tr_acc, tr_f1 = train_one_epoch(model, train_loader, optimizer, criterion, accumulation_steps)\n\n    # === VALIDATION\n    val_loss, val_acc, val_f1, val_preds, val_labels = eval_one_epoch(model, val_loader, criterion)\n\n    # === RECORD METRICS\n    train_losses.append(tr_loss)\n    val_losses.append(val_loss)\n    train_accs.append(tr_acc)\n    val_accs.append(val_acc)\n    train_f1s.append(tr_f1)\n    val_f1s.append(val_f1)\n\n    # === PRINT METRICS\n    print(f\"  🔧 Train Loss: {tr_loss:.4f} | Acc: {tr_acc:.4f} | F1: {tr_f1:.4f}\")\n    print(f\"  📉 Val   Loss: {val_loss:.4f} | Acc: {val_acc:.4f} | F1: {val_f1:.4f}\")\n\n    # === SAVE BEST MODEL BASED ON F1\n    if val_f1 > best_val_f1:\n        best_val_f1 = val_f1\n        epochs_no_improve = 0\n        print(\"  ✅ New best F1 found. Saving model...\")\n        torch.save(model.state_dict(), \"best_model_weights_ResNet101_F1.pth\")\n        torch.save(model, \"best_model_full_ResNet101_F1.pth\")\n    else:\n        epochs_no_improve += 1\n        print(f\"  ⏳ No improvement in F1 for {epochs_no_improve} epoch(s)\")\n\n    # === EARLY STOPPING\n    if epochs_no_improve >= early_stop_patience:\n        print(\"⛔ Early stopping triggered.\")\n        break\n\n    # # Free GPU memory\n    # torch.cuda.empty_cache()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.801883Z","iopub.execute_input":"2026-01-24T14:03:35.802366Z","iopub.status.idle":"2026-01-24T14:03:35.855217Z","shell.execute_reply.started":"2026-01-24T14:03:35.80234Z","shell.execute_reply":"2026-01-24T14:03:35.854145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torchvision import models\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\n\n\n# Loss, optimizer, scheduler\n# criterion = nn.CrossEntropyLoss(label_smoothing=0.1)\n# optimizer = optim.AdamW(model.parameters(), lr=LR, weight_decay=1e-4)\n# scheduler = ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=2)\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.855729Z","iopub.status.idle":"2026-01-24T14:03:35.85602Z","shell.execute_reply.started":"2026-01-24T14:03:35.855847Z","shell.execute_reply":"2026-01-24T14:03:35.855863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Train loss: {train_losses[-2]:.4f}, acc: {train_accs[-2]:.4f}\")\nprint(f\"Val   loss: {val_losses[-2]:.4f}, acc: {val_accs[-2]:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.857113Z","iopub.status.idle":"2026-01-24T14:03:35.857355Z","shell.execute_reply.started":"2026-01-24T14:03:35.857218Z","shell.execute_reply":"2026-01-24T14:03:35.857234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nmodel_loaded = torch.load(\"/kaggle/working/best_model_full_ResNet101_F1.pth\", weights_only=False)\nmodel_loaded = model_loaded.to(DEVICE)\n# model_loaded.eval()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.858316Z","iopub.status.idle":"2026-01-24T14:03:35.85861Z","shell.execute_reply.started":"2026-01-24T14:03:35.858458Z","shell.execute_reply":"2026-01-24T14:03:35.858471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom sklearn.metrics import confusion_matrix, classification_report\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\n\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel_loaded.to(DEVICE)\nmodel_loaded.eval()\n\n# Get image_ids from the validation DataFrame\nimage_ids_val = test_df[\"image_id\"].values\n\nval_preds = []\nval_labels = []\nval_image_ids = []\nval_loss_total = 0.0\nval_correct = 0\nval_samples = 0\n\ncriterion = torch.nn.CrossEntropyLoss()\n\ni = 0  # to index image_ids_val\n\nwith torch.no_grad():\n    for images, labels in val_loader:\n        images = images.to(DEVICE)\n        labels = labels.to(DEVICE)\n        \n        outputs = model_loaded(images)\n        loss = criterion(outputs, labels)\n        val_loss_total += loss.item() * images.size(0)\n\n        preds = torch.argmax(outputs, dim=1)\n\n        val_preds.extend(preds.cpu().numpy())\n        val_labels.extend(labels.cpu().numpy())\n        val_image_ids.extend(image_ids_val[i:i+len(images)])  # append batch of image_ids\n        i += len(images)\n\n        val_correct += (preds == labels).sum().item()\n        val_samples += labels.size(0)\n\n# Compute metrics\nval_loss_avg = val_loss_total / val_samples\nval_accuracy = val_correct / val_samples\n\nprint(f\"Val loss: {val_loss_avg:.4f}, Val accuracy: {val_accuracy:.4f}\")\n\n# Confusion matrix\ncm = confusion_matrix(val_labels, val_preds)\nprint(\"Classification Report:\")\nprint(classification_report(val_labels, val_preds, target_names=[\"Normal\", \"Abnormal\"]))\n\n# Plot\nplt.figure(figsize=(5, 5))\nplt.imshow(cm, cmap=\"Blues\")\nplt.title(\"Confusion Matrix\")\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"True\")\nplt.xticks([0, 1], [\"Normal\", \"Abnormal\"])\nplt.yticks([0, 1], [\"Normal\", \"Abnormal\"])\nfor i in range(2):\n    for j in range(2):\n        plt.text(j, i, cm[i, j], ha=\"center\", va=\"center\", color=\"red\")\nplt.tight_layout()\nplt.show()\n\n# Create DataFrame for all predictions\nresults_df = pd.DataFrame({\n    \"image_id\": val_image_ids,\n    \"true_label\": val_labels,\n    \"pred_label\": val_preds\n})\n\n# Filter wrong predictions\nwrong_df = results_df[results_df[\"true_label\"] != results_df[\"pred_label\"]]\nprint(\"\\nWrong Predictions:\")\nprint(wrong_df.head())\n\n# Save wrong predictions\n# wrong_df.to_csv(\"wrong_image_ids.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.859375Z","iopub.status.idle":"2026-01-24T14:03:35.859576Z","shell.execute_reply.started":"2026-01-24T14:03:35.85948Z","shell.execute_reply":"2026-01-24T14:03:35.859489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(wrong_df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.861281Z","iopub.status.idle":"2026-01-24T14:03:35.861598Z","shell.execute_reply.started":"2026-01-24T14:03:35.861491Z","shell.execute_reply":"2026-01-24T14:03:35.861502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Re-load predictions (if needed)\nwrong_df = pd.read_csv(\"wrong_image_ids.csv\")\n\n# Check duplicates in CSV before merging\ncsv_path = \"/kaggle/input/csv-data/train.csv\"\ndf = pd.read_csv(csv_path)\n\n# Drop duplicates for correct merge\ndf_unique = df[[\"image_id\", \"class_name\"]].drop_duplicates(\"image_id\")\n\n# Safe merge (avoid row multiplication)\nwrong_df = wrong_df.merge(df_unique, on=\"image_id\", how=\"left\")\n\n# Now group correctly\nwrong_normal = wrong_df[wrong_df[\"true_label\"] == 0]\nwrong_abnormal = wrong_df[wrong_df[\"true_label\"] == 1]\n\nprint(\"Wrong predictions total:\", len(wrong_df))\nprint(\"Wrong predictions — True Normal:\", len(wrong_normal))\nprint(\"Wrong predictions — True Abnormal:\", len(wrong_abnormal))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.862609Z","iopub.status.idle":"2026-01-24T14:03:35.863092Z","shell.execute_reply.started":"2026-01-24T14:03:35.862959Z","shell.execute_reply":"2026-01-24T14:03:35.862974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nfrom tqdm import tqdm\n\n# 1. Define your image source directory\nimage_folder = PNG_FOLDER  # change if your images are elsewhere\n\n# 2. Create output folders\nos.makedirs(\"wrong_normal_images\", exist_ok=True)\nos.makedirs(\"wrong_abnormal_images\", exist_ok=True)\n\n# 3. Copy images for wrong_normal\nfor image_id in tqdm(wrong_normal[\"image_id\"]):\n    src = os.path.join(image_folder, image_id + \".png\")  # adjust if it's .jpg or another format\n    dst = os.path.join(\"wrong_normal_images\", image_id + \".png\")\n    if os.path.exists(src):\n        shutil.copy(src, dst)\n\n# 4. Copy images for wrong_abnormal\nfor image_id in tqdm(wrong_abnormal[\"image_id\"]):\n    src = os.path.join(image_folder, image_id + \".png\")\n    dst = os.path.join(\"wrong_abnormal_images\", image_id + \".png\")\n    if os.path.exists(src):\n        shutil.copy(src, dst)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.86428Z","iopub.status.idle":"2026-01-24T14:03:35.865005Z","shell.execute_reply.started":"2026-01-24T14:03:35.864858Z","shell.execute_reply":"2026-01-24T14:03:35.864874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Define paths\nnormal_dir = \"wrong_normal_images\"\nabnormal_dir = \"wrong_abnormal_images\"\n\n# Count PNG files in each\nnormal_count = len([f for f in os.listdir(normal_dir) if f.endswith(\".png\")])\nabnormal_count = len([f for f in os.listdir(abnormal_dir) if f.endswith(\".png\")])\n\nprint(f\"🟢 wrong_normal_images folder: {normal_count} images\")\nprint(f\"🔴 wrong_abnormal_images folder: {abnormal_count} images\")\n\n# Optional: confirm they match the DataFrame lengths\nprint(f\"(Expected: {len(wrong_normal)} normal, {len(wrong_abnormal)} abnormal)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.865921Z","iopub.status.idle":"2026-01-24T14:03:35.866601Z","shell.execute_reply.started":"2026-01-24T14:03:35.866429Z","shell.execute_reply":"2026-01-24T14:03:35.866442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Zip the folders\n!zip -r wrong_normal_images.zip wrong_normal_images\n!zip -r wrong_abnormal_images.zip wrong_abnormal_images\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.867423Z","iopub.status.idle":"2026-01-24T14:03:35.867755Z","shell.execute_reply.started":"2026-01-24T14:03:35.867586Z","shell.execute_reply":"2026-01-24T14:03:35.867599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image\nimport os\nimport math\n\n# Path to image folder\nimages_folder = \"/kaggle/input/vinbigdata-chest-xray-resized-png-1024x1024/train\"\n\n# Total images to display\ntotal = len(wrong_normal)\ncols = 5\nrows = math.ceil(total / cols)\n\nplt.figure(figsize=(cols * 4, rows * 4))\nfor i, row in enumerate(wrong_normal.itertuples()):\n    img_path = os.path.join(images_folder, row.image_id + \".png\")\n    img = Image.open(img_path).convert(\"RGB\")\n    \n    plt.subplot(rows, cols, i + 1)\n    plt.imshow(img)\n    plt.title(f\"True: {row.class_name}\\nPred: Abnormal\", fontsize=9)\n    plt.axis(\"off\")\n\nplt.suptitle(\"Misclassified — True: Normal\", fontsize=18)\nplt.tight_layout(rect=[0, 0.03, 1, 0.95])\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.868745Z","iopub.status.idle":"2026-01-24T14:03:35.868978Z","shell.execute_reply.started":"2026-01-24T14:03:35.868869Z","shell.execute_reply":"2026-01-24T14:03:35.868881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport math\n\n# Folder containing images\nimages_folder = \"/kaggle/input/vinbigdata-chest-xray-resized-png-1024x1024/train\"\n\n# Number of images to plot\ntotal = len(wrong_abnormal)\ncols = 5\nrows = math.ceil(total / cols)\n\nplt.figure(figsize=(cols * 4, rows * 4))\n\nfor i, row in enumerate(wrong_abnormal.itertuples()):\n    img_path = os.path.join(images_folder, row.image_id + \".png\")\n    img = Image.open(img_path).convert(\"RGB\")\n\n    plt.subplot(rows, cols, i + 1)\n    plt.imshow(img)\n    plt.title(f\"True: {row.class_name}\\nPred: Normal\", fontsize=9)\n    plt.axis(\"off\")\n\nplt.suptitle(\"All Misclassified Images — True: Abnormal\", fontsize=18)\nplt.tight_layout(rect=[0, 0.03, 1, 0.95])\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.870093Z","iopub.status.idle":"2026-01-24T14:03:35.870411Z","shell.execute_reply.started":"2026-01-24T14:03:35.870245Z","shell.execute_reply":"2026-01-24T14:03:35.870258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nassert (wrong_abnormal[\"true_label\"] == 1).all()\n\n\nabnormal_error_counts = wrong_abnormal[\"class_name\"].value_counts().reset_index()\nabnormal_error_counts.columns = [\"class_name\", \"num_wrong_predictions\"]\n\n\nprint(abnormal_error_counts)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.87129Z","iopub.status.idle":"2026-01-24T14:03:35.871595Z","shell.execute_reply.started":"2026-01-24T14:03:35.871432Z","shell.execute_reply":"2026-01-24T14:03:35.871445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ncsv_path = \"/kaggle/input/csv-data/train.csv\"\ndf = pd.read_csv(csv_path)\ndf_unique = df[[\"image_id\", \"class_name\"]].drop_duplicates(\"image_id\")\n\n\nresults_df = results_df.merge(df_unique, on=\"image_id\", how=\"left\")\n\n# -----------------------------------\n\nwrong_df = results_df[results_df[\"true_label\"] != results_df[\"pred_label\"]]\nwrong_abnormal = wrong_df[wrong_df[\"true_label\"] == 1]\nwrong_counts = wrong_abnormal[\"class_name\"].value_counts().reset_index()\nwrong_counts.columns = [\"class_name\", \"num_wrong_predictions\"]\n\n# -----------------------------------\n\ncorrect_abnormal = results_df[\n    (results_df[\"true_label\"] == 1) & (results_df[\"pred_label\"] == 1)\n]\ncorrect_counts = correct_abnormal[\"class_name\"].value_counts().reset_index()\ncorrect_counts.columns = [\"class_name\", \"num_correct_predictions\"]\n\n# -----------------------------------\n\nsummary_df = pd.merge(correct_counts, wrong_counts, on=\"class_name\", how=\"outer\").fillna(0)\nsummary_df[[\"num_correct_predictions\", \"num_wrong_predictions\"]] = summary_df[[\"num_correct_predictions\", \"num_wrong_predictions\"]].astype(int)\n\n\nprint(summary_df.sort_values(by=\"class_name\").reset_index(drop=True))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.872328Z","iopub.status.idle":"2026-01-24T14:03:35.872628Z","shell.execute_reply.started":"2026-01-24T14:03:35.872478Z","shell.execute_reply":"2026-01-24T14:03:35.872493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load full dataset\ncsv_path = \"/kaggle/input/csv-data/train.csv\"\ndf = pd.read_csv(csv_path)\n\n\ndf_unique = df[[\"image_id\", \"class_name\"]].drop_duplicates(\"image_id\")\n\n\ntrain_ids = balanced_train_df[\"image_id\"].values\nval_ids = balanced_val_df[\"image_id\"].values\n\n\ndf_unique[\"set\"] = df_unique[\"image_id\"].apply(\n    lambda x: \"train\" if x in train_ids else \"val\" if x in val_ids else \"other\"\n)\n\npivot_table = pd.pivot_table(\n    df_unique,\n    index=\"class_name\",\n    columns=\"set\",\n    aggfunc=\"size\",\n    fill_value=0\n).reset_index()\n\n\npivot_table = pivot_table.rename(columns={\"train\": \"train_count\", \"val\": \"val_count\"})\n\n\nprint(pivot_table)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.873678Z","iopub.status.idle":"2026-01-24T14:03:35.873966Z","shell.execute_reply.started":"2026-01-24T14:03:35.873833Z","shell.execute_reply":"2026-01-24T14:03:35.873845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# ---- [1] Load Dataset + Unique image/class\ncsv_path = \"/kaggle/input/csv-data/train.csv\"\ndf = pd.read_csv(csv_path)\ndf_unique = df[[\"image_id\", \"class_name\"]].drop_duplicates(subset=\"image_id\")\n\n# ---- [2] Assign train / val to each image\ntrain_ids = set(balanced_train_df[\"image_id\"].values)\nval_ids = set(balanced_val_df[\"image_id\"].values)\n\ndf_unique[\"set\"] = df_unique[\"image_id\"].apply(\n    lambda x: \"train\" if x in train_ids else \"val\" if x in val_ids else \"other\"\n)\n\n# ---- [3] Compute train and val counts per class\npivot_table = pd.pivot_table(\n    df_unique,\n    index=\"class_name\",\n    columns=\"set\",\n    aggfunc=\"size\",\n    fill_value=0\n).reset_index()\n\npivot_table = pivot_table.rename(columns={\"train\": \"train_count\", \"val\": \"val_count\"})\n\n# ---- [4] Add class_name info to prediction results\nresults_df = results_df.merge(df_unique[[\"image_id\", \"class_name\"]], on=\"image_id\", how=\"left\")\n\n# ---- [5] Wrong predictions for abnormal class (true=1, pred=0)\nwrong_df = results_df[results_df[\"true_label\"] != results_df[\"pred_label\"]]\nwrong_abnormal = wrong_df[wrong_df[\"true_label\"] == 1]\nwrong_counts = wrong_abnormal[\"class_name\"].value_counts().reset_index()\nwrong_counts.columns = [\"class_name\", \"num_wrong_predictions\"]\n\n# ---- [6] Correct predictions for abnormal class (true=1, pred=1)\ncorrect_abnormal = results_df[\n    (results_df[\"true_label\"] == 1) & (results_df[\"pred_label\"] == 1)\n]\ncorrect_counts = correct_abnormal[\"class_name\"].value_counts().reset_index()\ncorrect_counts.columns = [\"class_name\", \"num_correct_predictions\"]\n\n# ---- [7] Merge correct + wrong counts\nperformance_df = pd.merge(correct_counts, wrong_counts, on=\"class_name\", how=\"outer\").fillna(0)\nperformance_df[[\"num_correct_predictions\", \"num_wrong_predictions\"]] = performance_df[\n    [\"num_correct_predictions\", \"num_wrong_predictions\"]\n].astype(int)\n\n# ---- [8] Merge with pivot_table to add train/val count\nsummary_df = pd.merge(pivot_table, performance_df, on=\"class_name\", how=\"outer\").fillna(0)\nsummary_df[[\"train_count\", \"val_count\", \"num_correct_predictions\", \"num_wrong_predictions\"]] = summary_df[\n    [\"train_count\", \"val_count\", \"num_correct_predictions\", \"num_wrong_predictions\"]\n].astype(int)\n\n# ---- [9] Final result\nsummary_df = summary_df.sort_values(\"class_name\").reset_index(drop=True)\nprint(summary_df)\n\n# ---- [Optional] Save to CSV\nsummary_df.to_csv(\"class_prediction_summary.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.874843Z","iopub.status.idle":"2026-01-24T14:03:35.87513Z","shell.execute_reply.started":"2026-01-24T14:03:35.874969Z","shell.execute_reply":"2026-01-24T14:03:35.874982Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**# Complete training on saved model**# ","metadata":{}},{"cell_type":"code","source":"# import torch\n# from torch import nn\n# from torch.utils.data import DataLoader\n# from sklearn.metrics import accuracy_score\n# from tqdm import tqdm\n# import timm\n\n# # ============ CONFIGURATION ============\n# DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n# LR = 5e-5\n# EPOCHS = 30  # Update this based on how many total epochs you want\n# start_epoch = 15  # Change this to the last saved epoch + 1\n# best_val_loss = float('inf')\n\n# # ============ LOAD SAVED MODEL ============\n# model_swin = torch.load(\"/kaggle/working/best_model_full_Swin_Model_.pth\" , weights_only=False)\n# model_swin = model_swin.to(DEVICE)\n  \n# # ============ LOSS AND OPTIMIZER ============\n# criterion = nn.CrossEntropyLoss()\n# optimizer = torch.optim.AdamW(model_swin.parameters(), lr=LR, weight_decay=1e-4)\n\n# # ============ TRAINING AND EVAL FUNCTIONS ============\n# def train_one_epoch(model, loader, optimizer, criterion):\n#     model.train()\n#     running_loss = 0\n#     all_preds, all_labels = [], []\n#     for imgs, labels in tqdm(loader):\n#         imgs, labels = imgs.to(DEVICE), labels.to(DEVICE)\n#         optimizer.zero_grad()\n#         outputs = model(imgs)\n#         loss = criterion(outputs, labels)\n#         loss.backward()\n#         optimizer.step()\n#         running_loss += loss.item() * imgs.size(0)\n#         preds = outputs.argmax(1).detach().cpu().numpy()\n#         all_preds.extend(preds)\n#         all_labels.extend(labels.cpu().numpy())\n#     epoch_loss = running_loss / len(loader.dataset)\n#     acc = accuracy_score(all_labels, all_preds)\n#     return epoch_loss, acc\n\n# def eval_one_epoch(model, loader, criterion):\n#     model.eval()\n#     running_loss = 0\n#     all_preds, all_labels = [], []\n#     with torch.no_grad():\n#         for imgs, labels in loader:\n#             imgs, labels = imgs.to(DEVICE), labels.to(DEVICE)\n#             outputs = model(imgs)\n#             loss = criterion(outputs, labels)\n#             running_loss += loss.item() * imgs.size(0)\n#             preds = outputs.argmax(1).detach().cpu().numpy()\n#             all_preds.extend(preds)\n#             all_labels.extend(labels.cpu().numpy())\n#     epoch_loss = running_loss / len(loader.dataset)\n#     acc = accuracy_score(all_labels, all_preds)\n#     return epoch_loss, acc, all_preds, all_labels\n\n\n# # ============ RESUME TRAINING ============\n# train_losses, val_losses, train_accs, val_accs = [], [], [], []\n\n# for epoch in range(start_epoch, EPOCHS):\n#     print(f\"Epoch {epoch+1}/{EPOCHS}\")\n    \n#     tr_loss, tr_acc = train_one_epoch(model_swin, train_loader, optimizer, criterion)\n#     val_loss, val_acc, val_preds, val_labels = eval_one_epoch(model_swin, val_loader, criterion)\n\n#     train_losses.append(tr_loss)\n#     val_losses.append(val_loss)\n#     train_accs.append(tr_acc)\n#     val_accs.append(val_acc)\n\n#     print(f\"  Train loss: {tr_loss:.4f}, acc: {tr_acc:.4f}\")\n#     print(f\"  Val   loss: {val_loss:.4f}, acc: {val_acc:.4f}\")\n\n#     if val_loss < best_val_loss:\n#         best_val_loss = val_loss\n#         print(\"  🔥 Best model so far. Saving...\")\n#         torch.save(model_swin.state_dict(), \"best_model_weights_Swin_Model_second_.pth\")\n#         torch.save(model_swin, \"best_model_full_Swin_Model_second_.pth\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.8759Z","iopub.status.idle":"2026-01-24T14:03:35.876207Z","shell.execute_reply.started":"2026-01-24T14:03:35.876045Z","shell.execute_reply":"2026-01-24T14:03:35.876058Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Loaded Model**","metadata":{}},{"cell_type":"code","source":"import torch\nfrom torchvision import transforms\nfrom PIL import Image\n\n# Set device\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# # Your normalization values used in training (example)\n# Mean: [0.54862876 0.54862876 0.54862876]\n# Std: [0.2669181 0.2669181 0.2669181]\n\n# Define transformations (same as used during training)\ntransform = transforms.Compose([\n    transforms.Resize((256, 256)),  # Adjust size if your model uses another input size\n    transforms.ToTensor(),\n    transforms.Normalize(mean=my_mean, std=my_std)\n])\n\n\n\n# Label map\nlabel_map = {0: \"Normal\", 1: \"Abnormal\"}\n\n# Function to load and preprocess image\ndef load_image(image_path):\n    img = Image.open(image_path).convert('RGB')\n    img = transform(img)\n    img = img.unsqueeze(0)  # add batch dimension\n    return img\n\n# Function to predict class from image path\ndef predict_image(model, image_path, device):\n    img_tensor = load_image(image_path).to(device)\n    with torch.no_grad():\n        output = model(img_tensor)\n        pred = torch.argmax(output, dim=1).item()\n    return pred\n\n# Example usage:\nimage_path = \"/kaggle/input/vinbigdata-chest-xray-resized-png-1024x1024/train/d3637a1935a905b3c326af31389cb846.png\"  # <-- change to your image path\n\npredicted_class = predict_image(model_loaded, image_path, DEVICE)\nprint(f\"Prediction: {label_map[predicted_class]}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.877797Z","iopub.status.idle":"2026-01-24T14:03:35.878083Z","shell.execute_reply.started":"2026-01-24T14:03:35.877928Z","shell.execute_reply":"2026-01-24T14:03:35.877942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport random\nfrom PIL import Image\n\n# Path to CSV and images folder\ntrain_csv_path = \"/kaggle/input/csv-data/train.csv\"  # adjust if needed\nimages_folder = \"/kaggle/input/vinbigdata-chest-xray-resized-png-1024x1024/train\"  # adjust if needed\n\ndf_train = pd.read_csv(train_csv_path)\n\n# Sample 30 random rows\nsample_df = df_train.sample(n=30).reset_index(drop=True)\n\ndef load_image(image_path):\n    img = Image.open(image_path).convert('RGB')\n    img_t = transform(img)# Transformed tensor (normalized, resized, etc.)\n    img_t = img_t.unsqueeze(0)\n    return img, img_t\n\ndef predict_image(model, img_tensor, device):\n    img_tensor = img_tensor.to(device)\n    with torch.no_grad():\n        output = model(img_tensor)\n        pred = torch.argmax(output, dim=1).item()\n    return pred\n\nplt.figure(figsize=(20, 15))\nfor i, row in sample_df.iterrows():\n    # Construct full image path\n    img_path = os.path.join(images_folder, row[\"image_id\"] + \".png\")  # add extension if missing\n\n    true_class_name = row[\"class_name\"]\n    true_binary_label = row[\"class_id\"]\n\n    # Load image & tensor\n    img, img_tensor = load_image(img_path)\n\n    # Predict use the loaded  model \n    pred_label = predict_image(model_loaded, img_tensor, DEVICE)\n    pred_class_name = label_map[pred_label]\n\n    plt.subplot(5, 6, i + 1)\n    plt.imshow(img)\n    plt.title(f\"True: {true_class_name}\\nPred: {pred_class_name}\", fontsize=9)\n    plt.axis(\"off\")\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.879162Z","iopub.status.idle":"2026-01-24T14:03:35.879394Z","shell.execute_reply.started":"2026-01-24T14:03:35.879298Z","shell.execute_reply":"2026-01-24T14:03:35.879307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# ------------ Display Wrong Normal Images ------------\nplt.figure(figsize=(18, 12))\nfor i, (_, row) in enumerate(wrong_normal.head(12).iterrows()):\n    img_path = os.path.join(images_folder, row[\"image_id\"] + \".png\")\n    img = Image.open(img_path).convert(\"RGB\")\n\n    plt.subplot(3, 4, i + 1)\n    plt.imshow(img)\n    plt.title(f\"True: Normal\\nPred: Abnormal\", fontsize=10)\n    plt.axis(\"off\")\n\nplt.suptitle(\"Wrong Predictions — True Label is Normal\", fontsize=16)\nplt.tight_layout(rect=[0, 0.03, 1, 0.95])\nplt.show()\n\n# ------------ Display Wrong Abnormal Images ------------\nplt.figure(figsize=(18, 12))\nfor i, (_, row) in enumerate(wrong_abnormal.head(12).iterrows()):\n    img_path = os.path.join(images_folder, row[\"image_id\"] + \".png\")\n    img = Image.open(img_path).convert(\"RGB\")\n\n    plt.subplot(3, 4, i + 1)\n    plt.imshow(img)\n    plt.title(f\"True: {row['class_name']}\\nPred: Normal\", fontsize=10)\n    plt.axis(\"off\")\n\nplt.suptitle(\"Wrong Predictions — True Label is Abnormal\", fontsize=16)\nplt.tight_layout(rect=[0, 0.03, 1, 0.95])\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.880411Z","iopub.status.idle":"2026-01-24T14:03:35.880672Z","shell.execute_reply.started":"2026-01-24T14:03:35.880511Z","shell.execute_reply":"2026-01-24T14:03:35.880521Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**LOAD MODEL AND EVALUATE**","metadata":{}},{"cell_type":"code","source":"import torch\nfrom PIL import Image\nfrom torchvision import transforms\nimport pydicom\nimport numpy as np\n\n# Set device\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Normalization values used in training\nmean = [0.54821104, 0.54821104, 0.54821104]\nstd = [0.26668723, 0.26668723, 0.26668723]\n\n# Label map\nlabel_map = {0: \"Normal\", 1: \"Abnormal\"}\n\n# Function to load full model\ndef load_full_model(model_path, device):\n    model = torch.load(model_path, map_location=device, weights_only=False)\n    model.eval()\n    return model\n\n# Convert DICOM image to PIL RGB\ndef dicom_to_pil(dicom_path):\n    ds = pydicom.dcmread(dicom_path)\n    img = ds.pixel_array.astype(np.float32)\n\n    # Normalize to [0, 255]\n    img -= img.min()\n    img /= (img.max() + 1e-6)\n    img *= 255.0\n    img = img.astype(np.uint8)\n\n    return Image.fromarray(img).convert(\"RGB\")\n\n# Predict from DICOM image path\ndef predict_dicom(model, dicom_path, mean, std, device):\n    transform = transforms.Compose([\n        transforms.Resize((256, 256)),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=mean, std=std)\n    ])\n\n    pil_img = dicom_to_pil(dicom_path)\n    img_tensor = transform(pil_img).unsqueeze(0).to(device)\n\n    with torch.no_grad():\n        output = model(img_tensor)\n        pred = torch.argmax(output, dim=1).item()\n\n    return pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.881543Z","iopub.status.idle":"2026-01-24T14:03:35.881825Z","shell.execute_reply.started":"2026-01-24T14:03:35.881658Z","shell.execute_reply":"2026-01-24T14:03:35.881668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#normal\n# \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/00053190460d56c53cc3e57321387478.dicom\"\n\n\n\n# Load model\nmodel_path = \"/kaggle/working/best_model_full_resnet101_NDL_.pth\"\nmodel_loaded = load_full_model(model_path, DEVICE)\n\n# Predict from DICOM\ndicom_path = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/0061cf6d35e253b6e7f03940592cc35e.dicom\"\npredicted_class = predict_dicom(model_loaded, dicom_path, mean, std, DEVICE)\n\nprint(f\"Prediction: {label_map[predicted_class]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:03:35.882863Z","iopub.status.idle":"2026-01-24T14:03:35.88314Z","shell.execute_reply.started":"2026-01-24T14:03:35.882998Z","shell.execute_reply":"2026-01-24T14:03:35.88301Z"}},"outputs":[],"execution_count":null}]}