{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":13224382,"datasetId":8382328,"databundleVersionId":13919222}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install -U ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T08:36:51.454165Z","iopub.execute_input":"2025-10-02T08:36:51.45467Z","iopub.status.idle":"2025-10-02T08:36:55.366303Z","shell.execute_reply.started":"2025-10-02T08:36:51.454643Z","shell.execute_reply":"2025-10-02T08:36:55.365513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Python cell\nfrom pathlib import Path\nimport os, random, time, json, math\nimport numpy as np, pandas as pd\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms, models\nimport torchvision.transforms.functional as TF\n\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score, classification_report, confusion_matrix\n\n# ultralytics\nfrom ultralytics import YOLO\n\n# global paths (update if needed)\nWORK_DIR = Path(\"/kaggle/working/vindr\")\nIMG_DIR = WORK_DIR / \"images\"\nLAB_DIR = WORK_DIR / \"labels\"\nMODEL_DIR = Path(\"/kaggle/working/models\")\nMODEL_DIR.mkdir(parents=True, exist_ok=True)\n\nTRAIN_CSV = Path(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv\")\nSAMPLE_SUB = Path(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/sample_submission.csv\")\nTEST_DICOM_DIR = Path(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/test\")\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(\"Device:\", device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T08:36:55.367199Z","iopub.execute_input":"2025-10-02T08:36:55.36748Z","iopub.status.idle":"2025-10-02T08:36:59.112743Z","shell.execute_reply.started":"2025-10-02T08:36:55.367447Z","shell.execute_reply":"2025-10-02T08:36:59.112121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install -U ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T08:36:59.113445Z","iopub.execute_input":"2025-10-02T08:36:59.113745Z","iopub.status.idle":"2025-10-02T08:37:02.728062Z","shell.execute_reply.started":"2025-10-02T08:36:59.113729Z","shell.execute_reply":"2025-10-02T08:37:02.727131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Python cell\nfrom pathlib import Path\nimport os, random, time, json, math\nimport numpy as np, pandas as pd\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms, models\nimport torchvision.transforms.functional as TF\n\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score, classification_report, confusion_matrix\n\n# ultralytics\nfrom ultralytics import YOLO\n\n# global paths (update if needed)\nWORK_DIR = Path(\"/kaggle/working/vindr\")\nIMG_DIR = WORK_DIR / \"images\"\nLAB_DIR = WORK_DIR / \"labels\"\nMODEL_DIR = Path(\"/kaggle/working/models\")\nMODEL_DIR.mkdir(parents=True, exist_ok=True)\n\nTRAIN_CSV = Path(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv\")\nSAMPLE_SUB = Path(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/sample_submission.csv\")\nTEST_DICOM_DIR = Path(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/test\")\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(\"Device:\", device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T08:37:02.730625Z","iopub.execute_input":"2025-10-02T08:37:02.730874Z","iopub.status.idle":"2025-10-02T08:37:02.738759Z","shell.execute_reply.started":"2025-10-02T08:37:02.730851Z","shell.execute_reply":"2025-10-02T08:37:02.738213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import shutil\n# import os\n\n# # Source (input) and destination (working/output)\n# src = \"/kaggle/input/chest-xray/kaggle/working/vindr\"       # change this to your dataset\n# dst = \"/kaggle/working/vindr\"     # destination folder\n\n# # If destination exists, remove it first (optional)\n# if os.path.exists(dst):\n#     shutil.rmtree(dst)\n\n# # Copy entire folder\n# shutil.copytree(src, dst)\n\n# print(f\"✅ Copied {src} → {dst}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T08:37:02.739637Z","iopub.execute_input":"2025-10-02T08:37:02.739884Z","iopub.status.idle":"2025-10-02T08:37:02.761039Z","shell.execute_reply.started":"2025-10-02T08:37:02.739862Z","shell.execute_reply":"2025-10-02T08:37:02.760471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Build class list (match earlier)\nclass_names = [\n    \"Aortic_enlargement\",\"Atelectasis\",\"Calcification\",\"Cardiomegaly\",\n    \"Consolidation\",\"ILD\",\"Infiltration\",\"Lung_Opacity\",\"Nodule_Mass\",\n    \"Other_lesion\",\"Pleural_effusion\",\"Pleural_thickening\",\n    \"Pneumothorax\",\"Pulmonary_fibrosis\",\"No_finding\"\n]\nNUM_CLASSES = len(class_names)\n\n# Read train.csv and build multi-hot label dict\ndf = pd.read_csv(TRAIN_CSV)\n# group by image_id\ntargets = {}\nfor img_id, g in df.groupby(\"image_id\"):\n    vec = np.zeros(NUM_CLASSES, dtype=np.float32)\n    for cid in g['class_id'].values:\n        vec[int(cid)] = 1.0\n    targets[img_id] = vec\n\n# Some images may be missing in df (no annotation) -> treat as No_finding\n# Ensure that for converted images, we have label vectors\ntrain_img_dir = IMG_DIR / \"train\"\nval_img_dir   = IMG_DIR / \"val\"\n\ntrain_ids = [p.stem for p in train_img_dir.glob(\"*.jpg\")]\nval_ids   = [p.stem for p in val_img_dir.glob(\"*.jpg\")]\n\n# if an image not in targets -> treat as no finding (class 14 = No_finding)\nfor img in train_ids + val_ids:\n    if img not in targets:\n        vec = np.zeros(NUM_CLASSES, dtype=np.float32)\n        vec[14] = 1.0\n        targets[img] = vec\n\n# Dataset class\nclass MultiLabelCXRDataset(Dataset):\n    def __init__(self, image_dir, img_ids, targets_dict, transform=None):\n        self.image_dir = Path(image_dir)\n        self.img_ids = img_ids\n        self.targets = targets_dict\n        self.transform = transform\n\n    def __len__(self): return len(self.img_ids)\n\n    def __getitem__(self, idx):\n        img_id = self.img_ids[idx]\n        img_path = self.image_dir / f\"{img_id}.jpg\"\n        img = Image.open(img_path).convert(\"RGB\")\n        if self.transform:\n            img = self.transform(img)\n        label = torch.tensor(self.targets[img_id], dtype=torch.float32)\n        return img, label, img_id\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T08:37:02.761886Z","iopub.execute_input":"2025-10-02T08:37:02.76206Z","iopub.status.idle":"2025-10-02T08:37:03.461956Z","shell.execute_reply.started":"2025-10-02T08:37:02.762046Z","shell.execute_reply":"2025-10-02T08:37:03.461382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BATCH = 32\n\ntrain_tfms = transforms.Compose([\n    transforms.Resize((224,224)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(5),\n    transforms.ColorJitter(brightness=0.1, contrast=0.1),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485,0.456,0.406],[0.229,0.224,0.225])\n])\n\nval_tfms = transforms.Compose([\n    transforms.Resize((224,224)),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485,0.456,0.406],[0.229,0.224,0.225])\n])\n\ntrain_ds = MultiLabelCXRDataset(train_img_dir, train_ids, targets, transform=train_tfms)\nval_ds   = MultiLabelCXRDataset(val_img_dir, val_ids, targets, transform=val_tfms)\n\ntrain_loader = DataLoader(train_ds, batch_size=BATCH, shuffle=True, num_workers=4, pin_memory=True)\nval_loader   = DataLoader(val_ds, batch_size=BATCH, shuffle=False, num_workers=4, pin_memory=True)\nprint(\"Train / Val sizes:\", len(train_ds), len(val_ds))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T08:37:03.462704Z","iopub.execute_input":"2025-10-02T08:37:03.462975Z","iopub.status.idle":"2025-10-02T08:37:03.470778Z","shell.execute_reply.started":"2025-10-02T08:37:03.462956Z","shell.execute_reply":"2025-10-02T08:37:03.470031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torchvision.models import resnet50, ResNet50_Weights\nfrom torch.amp import autocast, GradScaler\n\n# Model\nNUM_CLASSES = 15  # change as needed\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nmodel_cls = resnet50(weights=ResNet50_Weights.IMAGENET1K_V1)\nmodel_cls.fc = nn.Linear(model_cls.fc.in_features, NUM_CLASSES)\nmodel_cls = model_cls.to(device)\n\n# Loss, optimizer, scheduler\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.AdamW(model_cls.parameters(), lr=1e-4, weight_decay=1e-5)\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=2)\n\n# AMP\nscaler = GradScaler(\"cuda\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T08:37:03.471521Z","iopub.execute_input":"2025-10-02T08:37:03.4717Z","iopub.status.idle":"2025-10-02T08:37:04.147425Z","shell.execute_reply.started":"2025-10-02T08:37:03.471686Z","shell.execute_reply":"2025-10-02T08:37:04.146852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nlabels_dir = \"/kaggle/working/vindr/labels\"\n\nfor split in [\"train\", \"val\", \"test\"]:\n    folder = os.path.join(labels_dir, split)\n    for file in os.listdir(folder):\n        if file.endswith(\".txt\"):\n            path = os.path.join(folder, file)\n            with open(path, \"r\") as f:\n                lines = f.readlines()\n            \n            new_lines = []\n            for line in lines:\n                parts = line.strip().split()\n                if len(parts) >= 5:\n                    # force all labels to class 0\n                    parts[0] = \"0\"\n                    new_lines.append(\" \".join(parts) + \"\\n\")\n            \n            # overwrite\n            with open(path, \"w\") as f:\n                f.writelines(new_lines)\n\nprint(\"✅ All labels remapped to single class 0\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T08:37:04.14817Z","iopub.execute_input":"2025-10-02T08:37:04.148387Z","iopub.status.idle":"2025-10-02T08:37:05.606451Z","shell.execute_reply.started":"2025-10-02T08:37:04.14837Z","shell.execute_reply":"2025-10-02T08:37:05.605814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import yaml\nyolo_yaml = {\n    \"path\": str(WORK_DIR),   # base path\n    \"train\": \"images/train\",\n    \"val\":   \"images/val\",\n    \"test\":  \"images/test\",\n    \"nc\": 1,\n    \"names\": [\"abnormal\"]\n}\nYAML_PATH = WORK_DIR / \"yolov8_abnormal.yaml\"\nwith open(YAML_PATH, 'w') as f:\n    yaml.dump(yolo_yaml, f)\nprint(\"Saved YAML:\", YAML_PATH)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T08:37:05.607294Z","iopub.execute_input":"2025-10-02T08:37:05.607827Z","iopub.status.idle":"2025-10-02T08:37:05.613662Z","shell.execute_reply.started":"2025-10-02T08:37:05.607798Z","shell.execute_reply":"2025-10-02T08:37:05.612969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install -U ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T08:37:05.614305Z","iopub.execute_input":"2025-10-02T08:37:05.614542Z","iopub.status.idle":"2025-10-02T08:37:09.20029Z","shell.execute_reply.started":"2025-10-02T08:37:05.614527Z","shell.execute_reply":"2025-10-02T08:37:09.199473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\n\nyolo_model = YOLO(\"yolov8n.pt\")   # lightweight test model\nresults = yolo_model.train(\n    data=str(YAML_PATH),\n    epochs=20,\n    imgsz=640,\n    batch=8,\n    project=str(WORK_DIR/\"yolov8_runs\"),\n    name=\"abnormal_singleclass\"\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T08:37:09.201252Z","iopub.execute_input":"2025-10-02T08:37:09.201553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}