{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594,"isSourceIdPinned":false}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"intro_md","cell_type":"markdown","source":"# 🫁 CliniScan: Unified Milestone 1 & 2\n## Lung Abnormality Detection on Chest X-Rays\n\n**Project:** CliniScan  \n**Dataset:** VinBigData Chest X-ray Abnormalities Detection (Kaggle)  \n**Environment:** Kaggle Notebooks with GPU (P100 or T4x2 recommended)  \n\n---\n\n### Objectives Combined\n- **Milestone 1**: Data Preprocessing & Exploratory Data Analysis (EDA)\n- **Milestone 2**: Baseline Model Training (Classification & Detection)\n\n### Models Chosen\n- **Classification:** EfficientNet-B0 (Global diagnosis)\n- **Detection:** YOLOv8s (Local bounding boxes)\n","metadata":{}},{"id":"setup_md","cell_type":"markdown","source":"---\n## 📦 Section 1: Environment Setup & Library Imports\nInstall required libraries for DICOM processing, augmentations, and YOLOv8.","metadata":{}},{"id":"install_libs","cell_type":"code","source":"# 1.1 Install dependencies\n!pip install -q pydicom opencv-python-headless albumentations scikit-learn \\\n               matplotlib seaborn tqdm Pillow ultralytics\n\nprint('✅ All packages installed successfully.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T14:33:02.897375Z","iopub.execute_input":"2026-03-13T14:33:02.898136Z","iopub.status.idle":"2026-03-13T14:33:06.3836Z","shell.execute_reply.started":"2026-03-13T14:33:02.898106Z","shell.execute_reply":"2026-03-13T14:33:06.38261Z"}},"outputs":[],"execution_count":null},{"id":"import_libs","cell_type":"code","source":"# 1.2 Import necessary libraries\nimport os\nimport shutil\nimport glob\nimport warnings\nimport random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport seaborn as sns\nimport cv2\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom tqdm.auto import tqdm\nfrom sklearn.model_selection import train_test_split\nimport torch\nimport torch.nn as nn\nfrom torchvision import models, transforms\nfrom torch.utils.data import DataLoader, Dataset\nfrom ultralytics import YOLO\n\nwarnings.filterwarnings('ignore')\n# Set seeds for reproducibility\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\nif torch.cuda.is_available():\n    torch.cuda.manual_seed(SEED)\n    torch.cuda.manual_seed_all(SEED)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nprint('✅ Libraries imported and random seeds set.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T14:33:06.385335Z","iopub.execute_input":"2026-03-13T14:33:06.385613Z","iopub.status.idle":"2026-03-13T14:33:06.395195Z","shell.execute_reply.started":"2026-03-13T14:33:06.385588Z","shell.execute_reply":"2026-03-13T14:33:06.394515Z"}},"outputs":[],"execution_count":null},{"id":"check_gpu","cell_type":"code","source":"# 1.3 Verify GPU Availability\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"🚀 Using device: {device}\")\nif device.type == 'cuda':\n    print(torch.cuda.get_device_name(0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T14:33:06.39613Z","iopub.execute_input":"2026-03-13T14:33:06.396459Z","iopub.status.idle":"2026-03-13T14:33:06.412021Z","shell.execute_reply.started":"2026-03-13T14:33:06.396433Z","shell.execute_reply":"2026-03-13T14:33:06.411198Z"}},"outputs":[],"execution_count":null},{"id":"paths_md","cell_type":"markdown","source":"---\n## 📂 Section 2: Data Paths & EDA\nDefine Kaggle Dataset paths and perform basic Exploratory Data Analysis.","metadata":{}},{"id":"define_paths","cell_type":"code","source":"# 2.1 Define Paths\n# Input Dataset (Kaggle Specific)\nBASE_DIR   = '/kaggle/input/vinbigdata-chest-xray-abnormalities-detection'\nTRAIN_DIR  = os.path.join(BASE_DIR, 'train')\nTEST_DIR   = os.path.join(BASE_DIR, 'test')\nCSV_PATH   = os.path.join(BASE_DIR, 'train.csv')\n\n# Output Workspace\nOUTPUT_DIR = '/kaggle/working/cliniscan'\nIMG_TRAIN  = os.path.join(OUTPUT_DIR, 'dataset/images/train')\nIMG_VAL    = os.path.join(OUTPUT_DIR, 'dataset/images/val')\nLBL_TRAIN  = os.path.join(OUTPUT_DIR, 'dataset/labels/train')\nLBL_VAL    = os.path.join(OUTPUT_DIR, 'dataset/labels/val')\n\n# Create directories\nfor d in [IMG_TRAIN, IMG_VAL, LBL_TRAIN, LBL_VAL]:\n    os.makedirs(d, exist_ok=True)\n\ntrain_dicoms = sorted(glob.glob(os.path.join(TRAIN_DIR, '*.dicom')))\nprint(f'📁 Train DICOM files found: {len(train_dicoms):,}')\nif len(train_dicoms) == 0:\n    # Fallback to alternative kaggle path if needed\n    BASE_DIR = '/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection'\n    TRAIN_DIR  = os.path.join(BASE_DIR, 'train')\n    CSV_PATH   = os.path.join(BASE_DIR, 'train.csv')\n    train_dicoms = sorted(glob.glob(os.path.join(TRAIN_DIR, '*.dicom')))\n    print(f'📁 Fallback: Train DICOM files found: {len(train_dicoms):,}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T14:33:06.413782Z","iopub.execute_input":"2026-03-13T14:33:06.413997Z","iopub.status.idle":"2026-03-13T14:33:07.768594Z","shell.execute_reply.started":"2026-03-13T14:33:06.413978Z","shell.execute_reply":"2026-03-13T14:33:07.767956Z"}},"outputs":[],"execution_count":null},{"id":"load_csv","cell_type":"code","source":"# 2.2 Load and Analyze Annotations\ntry:\n    df = pd.read_csv(CSV_PATH)\n    print(f'📊 Annotation CSV shape: {df.shape}')\n    \n    print('\\n📌 DATASET SUMMARY')\n    print(f'  Unique images (train) : {df[\"image_id\"].nunique():,}')\n    print(f'  Total annotations     : {len(df):,}')\n    print(f'  Unique class labels   : {df[\"class_name\"].nunique()}')\n    \n    print('\\nClass distribution:')\n    print(df['class_name'].value_counts())\nexcept FileNotFoundError:\n    print(\"⚠️ CSV file not found. Please ensure the Kaggle dataset is attached.\")\n    # Create a dummy dataframe for testing if file missing\n    df = pd.DataFrame({'image_id': [], 'class_name': [], 'class_id': []})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T14:33:07.769437Z","iopub.execute_input":"2026-03-13T14:33:07.76971Z","iopub.status.idle":"2026-03-13T14:33:07.990731Z","shell.execute_reply.started":"2026-03-13T14:33:07.769688Z","shell.execute_reply":"2026-03-13T14:33:07.990156Z"}},"outputs":[],"execution_count":null},{"id":"prep_md","cell_type":"markdown","source":"---\n## ⚙️ Section 3: Data Preprocessing (DICOM -> PNG)\nConvert DICOM images to PNG, apply VOI LUT windowing, and handle monochrome photometric interpretations.","metadata":{}},{"id":"process_dicom_func","cell_type":"code","source":"# 3.1 DICOM Processing Function\nIMG_SIZE = 512 # Resizing to 512x512 for baseline speed\n\ndef process_dicom(path, img_size=IMG_SIZE):\n    \"\"\"\n    Reads a DICOM file, applies VOI LUT if present, handles Monochrome1/2,\n    normalizes to 8-bit, and resizes.\n    \"\"\"\n    dicom = pydicom.dcmread(path)\n    # Apply VOI LUT (Window Center/Width)\n    data = apply_voi_lut(dicom.pixel_array, dicom)\n    \n    # Handle Monochrome1 (invert so bones are white)\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    # Normalize to 0-255\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    \n    # Resize\n    data_resized = cv2.resize(data, (img_size, img_size))\n    return data_resized","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T14:33:07.991676Z","iopub.execute_input":"2026-03-13T14:33:07.991965Z","iopub.status.idle":"2026-03-13T14:33:07.997105Z","shell.execute_reply.started":"2026-03-13T14:33:07.991928Z","shell.execute_reply":"2026-03-13T14:33:07.996555Z"}},"outputs":[],"execution_count":null},{"id":"run_conversion","cell_type":"code","source":"# 3.2 Run Conversion on a Subset (Sanity Check / Baseline)\n# For a full run, set NUM_IMAGES = len(df['image_id'].unique())\nNUM_IMAGES = 500 \n\nif not df.empty:\n    sample_ids = df['image_id'].unique()[:NUM_IMAGES]\n    print(f\"🚀 Starting DICOM to PNG conversion for {len(sample_ids)} images...\")\n    \n    for img_id in tqdm(sample_ids):\n        img_path = os.path.join(TRAIN_DIR, f\"{img_id}.dicom\")\n        if os.path.exists(img_path):\n            img_array = process_dicom(img_path)\n            # Save to train directory for now (split later if needed)\n            out_path = os.path.join(IMG_TRAIN, f\"{img_id}.png\")\n            cv2.imwrite(out_path, img_array)\n    print(\"✅ Conversion complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-13T14:33:07.997882Z","iopub.execute_input":"2026-03-13T14:33:07.998152Z"}},"outputs":[],"execution_count":null},{"id":"yolo_md","cell_type":"markdown","source":"---\n## 🎯 Section 4: Detection Model (YOLOv8)\nSetup YOLO dataset configuration and initialize the baseline detection model.","metadata":{}},{"id":"yolo_config","cell_type":"code","source":"# 4.1 Create data.yaml for YOLOv8\nyaml_content = f\"\"\"\npath: {OUTPUT_DIR}/dataset\ntrain: images/train\nval: images/train  # Using train as val for dummy demonstration. Split properly in full run.\nnc: 14\nnames: ['Aortic enlargement', 'Atelectasis', 'Calcification', 'Cardiomegaly', 'Consolidation', 'ILD', 'Infiltration', 'Lung Opacity', 'Nodule/Mass', 'Other lesion', 'Pleural effusion', 'Pleural thickening', 'Pneumothorax', 'Pulmonary fibrosis']\n\"\"\"\n\nyaml_path = os.path.join(OUTPUT_DIR, \"data.yaml\")\nwith open(yaml_path, 'w') as f: \n    f.write(yaml_content)\n    \nprint(f\"✅ YOLO data.yaml created at {yaml_path}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"yolo_train","cell_type":"code","source":"# 4.2 Initialize Detection Model\nmodel_det = YOLO('yolov8s.pt')\nprint(\"✅ YOLOv8s Model Initialized.\")\n\n# 4.3 Training Command (Uncomment to train)\n# print(\"🚀 Starting YOLO Training...\")\n# results = model_det.train(data=yaml_path, epochs=10, imgsz=512, batch=16, device=device)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"effnet_md","cell_type":"markdown","source":"---\n## 🩺 Section 5: Classification Model (EfficientNet-B0)\nTrain a classifier to predict 'Abnormal' vs 'No finding'.","metadata":{}},{"id":"effnet_setup","cell_type":"code","source":"# 5.1 Define PyTorch Dataset\nclass CliniScanDataset(Dataset):\n    def __init__(self, df, img_dir, transform=None):\n        self.img_dir = img_dir\n        self.transform = transform\n        \n        # Create a unique list of image IDs from the dataframe\n        # For this baseline, we only use the ones we converted\n        converted_files = [f.replace('.png','') for f in os.listdir(img_dir)]\n        self.img_ids = [idx for idx in df['image_id'].unique() if idx in converted_files]\n        \n        # Label: 0 if 'No finding', 1 if Abnormal (any other class)\n        # If an image has multiple rows, we aggregate. If all are 'No finding' -> 0, else 1.\n        self.labels = {}\n        for img_id in self.img_ids:\n            subset = df[df['image_id'] == img_id]\n            if all(subset['class_name'] == 'No finding'):\n                self.labels[img_id] = 0\n            else:\n                self.labels[img_id] = 1\n\n    def __len__(self):\n        return len(self.img_ids)\n\n    def __getitem__(self, idx):\n        img_id = self.img_ids[idx]\n        img_path = os.path.join(self.img_dir, f\"{img_id}.png\")\n        \n        # Read image\n        img = cv2.imread(img_path)\n        if img is None:\n            # Fallback black image if loading fails\n            img = np.zeros((512, 512, 3), dtype=np.uint8)\n        else:\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n            \n        label = self.labels[img_id]\n        \n        if self.transform:\n            img = self.transform(img)\n            \n        return img, torch.tensor(label, dtype=torch.long)\n\nprint(\"✅ Dataset class defined.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"effnet_train","cell_type":"code","source":"# 5.2 Initialize Model and DataLoaders\ntrain_transform = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.Resize((256, 256)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\nif not df.empty and len(os.listdir(IMG_TRAIN)) > 0:\n    train_dataset = CliniScanDataset(df, IMG_TRAIN, transform=train_transform)\n    train_loader = DataLoader(train_dataset, batch_size=16, shuffle=True)\n    print(f\"Loaded {len(train_dataset)} images for classification training.\")\n\n# Load EfficientNet-B0\nmodel_cls = models.efficientnet_b0(weights=models.EfficientNet_B0_Weights.DEFAULT)\n# Modify the final fully connected layer for binary classification\nin_features = model_cls.classifier[1].in_features\nmodel_cls.classifier[1] = nn.Linear(in_features, 2)\nmodel_cls.to(device)\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model_cls.parameters(), lr=1e-4)\n\nprint(f\"✅ Classification Model (EfficientNet-B0) Ready on {device}\")\n\n# 5.3 Training Loop (Uncomment to train)\n\nepochs = 5\nmodel_cls.train()\nfor epoch in range(epochs):\n    running_loss = 0.0\n    for i, (inputs, targets) in enumerate(train_loader):\n        inputs, targets = inputs.to(device), targets.to(device)\n        \n        optimizer.zero_grad()\n        outputs = model_cls(inputs)\n        loss = criterion(outputs, targets)\n        loss.backward()\n        optimizer.step()\n        \n        running_loss += loss.item()\n        \n    print(f\"Epoch [{(epoch+1)}/{epochs}] Loss: {running_loss/len(train_loader):.4f}\")\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}