{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594}],"dockerImageVersionId":31287,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🫁 CliniScan – Lung Abnormality Detection on Chest X-Rays\n## Milestone 1: Data Preparation & Initial Setup\n\n**Project:** CliniScan  \n**Dataset:** VinBigData Chest X-ray Abnormalities Detection (Kaggle)  \n**Environment:** Kaggle Notebooks with GPU  \n\n---\n\n### Milestone 1 Objectives\n- ✅ Set up the development environment\n- ✅ Explore and understand the VinDr-CXR dataset\n- ✅ Convert DICOM images → PNG/JPEG\n- ✅ Parse and convert annotations (CSV → YOLO format)\n- ✅ Organize data into train / validation splits\n- ✅ Visualize samples and verify correctness\n\n---\n\n### Dataset Structure (Kaggle Competition)\n```\n/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/\n    train/          <- DICOM images for training\n    test/           <- DICOM images for testing\n    train.csv       <- Annotations (bounding boxes + class labels)\n    sample_submission.csv\n```\n\n### 14 Disease Classes\n```\n0: Aortic enlargement   1: Atelectasis        2: Calcification\n3: Cardiomegaly         4: Consolidation      5: ILD\n6: Infiltration         7: Lung Opacity       8: Nodule/Mass\n9: Other lesion        10: Pleural effusion  11: Pleural thickening\n12: Pneumothorax       13: Pulmonary fibrosis 14: No finding\n```","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## 📦 Section 1: Environment Setup\nInstall all required libraries and verify GPU availability.","metadata":{}},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n     for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T09:47:13.398776Z","iopub.execute_input":"2026-03-14T09:47:13.399053Z","iopub.status.idle":"2026-03-14T09:48:36.691363Z","shell.execute_reply.started":"2026-03-14T09:47:13.399021Z","shell.execute_reply":"2026-03-14T09:48:36.6905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 1.1  Install dependencies ──────────────────────────────────────────────\n!pip install -q pydicom opencv-python-headless albumentations scikit-learn \\\n               matplotlib seaborn tqdm Pillow\n\nprint('✅ All packages installed successfully.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T09:48:36.693332Z","iopub.execute_input":"2026-03-14T09:48:36.693699Z","iopub.status.idle":"2026-03-14T09:48:42.141794Z","shell.execute_reply.started":"2026-03-14T09:48:36.693674Z","shell.execute_reply":"2026-03-14T09:48:42.14081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 1.2  Import libraries ──────────────────────────────────────────────────\nimport os\nimport shutil\nimport glob\nimport warnings\nimport random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport matplotlib.gridspec as gridspec\nimport seaborn as sns\nimport cv2\nimport pydicom\nfrom PIL import Image\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nfrom sklearn.model_selection import train_test_split\n\nwarnings.filterwarnings('ignore')\nrandom.seed(42)\nnp.random.seed(42)\n\nprint('✅ Libraries imported.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T09:48:42.143147Z","iopub.execute_input":"2026-03-14T09:48:42.143503Z","iopub.status.idle":"2026-03-14T09:48:44.9826Z","shell.execute_reply.started":"2026-03-14T09:48:42.143474Z","shell.execute_reply":"2026-03-14T09:48:44.982027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 1.3  Verify GPU ────────────────────────────────────────────────────────\nimport subprocess\ntry:\n    gpu_info = subprocess.check_output(['nvidia-smi'], stderr=subprocess.DEVNULL).decode()\n    print('🚀 GPU is available!')\n    print(gpu_info)\nexcept Exception:\n    print('⚠️  No GPU detected. Go to: Notebook Settings → Accelerator → GPU')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## 📂 Section 2: Dataset Overview & Exploration","metadata":{}},{"cell_type":"code","source":"# ── 2.1  Define paths ──────────────────────────────────────────────────────\n# ── 2.1  Define paths ──────────────────────────────────────────────────────\nBASE_DIR   = '/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection'\nTRAIN_DIR  = os.path.join(BASE_DIR, 'train')\nTEST_DIR   = os.path.join(BASE_DIR, 'test')\nCSV_PATH   = os.path.join(BASE_DIR, 'train.csv')\n\n# Output workspace\nOUTPUT_DIR = '/kaggle/working/cliniscan'\nIMG_TRAIN  = os.path.join(OUTPUT_DIR, 'dataset/images/train')\nIMG_VAL    = os.path.join(OUTPUT_DIR, 'dataset/images/val')\nLBL_TRAIN  = os.path.join(OUTPUT_DIR, 'dataset/labels/train')\nLBL_VAL    = os.path.join(OUTPUT_DIR, 'dataset/labels/val')\n\nfor d in [IMG_TRAIN, IMG_VAL, LBL_TRAIN, LBL_VAL]:\n    os.makedirs(d, exist_ok=True)\n\ntrain_dicoms = sorted(glob.glob(os.path.join(TRAIN_DIR, '*.dicom')))\ntest_dicoms  = sorted(glob.glob(os.path.join(TEST_DIR, '*.dicom')))\nprint(f'📁 Train DICOM files : {len(train_dicoms):,}')\nprint(f'📁 Test  DICOM files : {len(test_dicoms):,}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 2.2  Load & preview annotations CSV ───────────────────────────────────\ndf = pd.read_csv(CSV_PATH)\nprint(f'📊 Annotation CSV shape: {df.shape}')\nprint(f'\\nColumns: {list(df.columns)}')\ndf.head(10)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 2.3  Basic statistics ──────────────────────────────────────────────────\nprint('='*55)\nprint('📌 DATASET SUMMARY')\nprint('='*55)\nprint(f'  Unique images (train) : {df[\"image_id\"].nunique():,}')\nprint(f'  Total annotations     : {len(df):,}')\nprint(f'  Unique class labels   : {df[\"class_name\"].nunique()}')\nprint(f'  Unique radiologists   : {df[\"rad_id\"].nunique()}')\nprint()\nprint('Class distribution:')\nprint(df['class_name'].value_counts().to_string())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 2.4  EDA – Class distribution chart ──────────────────────────────────\nfig, axes = plt.subplots(1, 2, figsize=(18, 6))\nfig.suptitle('VinDr-CXR Dataset – Exploratory Data Analysis', fontsize=16, fontweight='bold')\n\n# Bar chart\nclass_counts = df['class_name'].value_counts()\ncolors = plt.cm.tab20(np.linspace(0, 1, len(class_counts)))\naxes[0].barh(class_counts.index, class_counts.values, color=colors)\naxes[0].set_xlabel('Count', fontsize=12)\naxes[0].set_title('Annotation Count per Class', fontsize=13, fontweight='bold')\naxes[0].invert_yaxis()\nfor i, v in enumerate(class_counts.values):\n    axes[0].text(v + 50, i, str(v), va='center', fontsize=9)\n\n# Normal vs Abnormal pie\nno_finding = (df['class_name'] == 'No finding').sum()\nabnormal   = len(df) - no_finding\naxes[1].pie([no_finding, abnormal],\n            labels=['No Finding', 'Abnormal'],\n            autopct='%1.1f%%',\n            colors=['#4CAF50', '#F44336'],\n            startangle=90,\n            textprops={'fontsize': 13})\naxes[1].set_title('Normal vs Abnormal Annotations', fontsize=13, fontweight='bold')\n\nplt.tight_layout()\nplt.savefig(f'{OUTPUT_DIR}/eda_class_distribution.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint('✅ EDA chart saved.')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 2.5  Inspect a single DICOM file ──────────────────────────────────────\nsample_dcm_path = train_dicoms[0]\nsample_dcm = pydicom.dcmread(sample_dcm_path)\n\nprint('🔍 DICOM Metadata (Sample File)')\nprint(f'  File          : {os.path.basename(sample_dcm_path)}')\nprint(f'  Rows (Height) : {sample_dcm.Rows}')\nprint(f'  Cols (Width)  : {sample_dcm.Columns}')\nprint(f'  Pixel Array   : dtype={sample_dcm.pixel_array.dtype}, shape={sample_dcm.pixel_array.shape}')\nprint(f'  Min / Max px  : {sample_dcm.pixel_array.min()} / {sample_dcm.pixel_array.max()}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## 🔄 Section 3: DICOM → PNG Conversion\nWe apply VOI LUT windowing to get correct contrast, then normalize pixel values to 8-bit [0, 255] and resize to **1024 × 1024**.","metadata":{}},{"cell_type":"code","source":"# ── 3.1  Conversion function ───────────────────────────────────────────────\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\nTARGET_SIZE = (1024, 1024)   # resize target\n\ndef dicom_to_png(dicom_path: str, out_path: str, size: tuple = TARGET_SIZE) -> tuple:\n    \"\"\"\n    Convert a single DICOM file to a normalized 8-bit PNG.\n\n    Steps:\n      1. Read DICOM with pydicom.\n      2. Apply VOI LUT windowing (if available) for proper contrast.\n      3. Normalize pixel array to [0, 255] uint8.\n      4. Resize with OpenCV (INTER_LINEAR for upscale, INTER_AREA for downscale).\n      5. Save as PNG.\n\n    Returns: (original_height, original_width)\n    \"\"\"\n    dcm = pydicom.dcmread(dicom_path)\n    img = dcm.pixel_array.astype(np.float32)\n\n    # Apply VOI LUT for correct window-level/width\n    try:\n        img = apply_voi_lut(img, dcm)\n    except Exception:\n        pass  # fall back to raw pixel array if no VOI LUT\n\n    # Handle MONOCHROME1 (invert so lungs appear dark on white)\n    if hasattr(dcm, 'PhotometricInterpretation'):\n        if dcm.PhotometricInterpretation == 'MONOCHROME1':\n            img = img.max() - img\n\n    orig_h, orig_w = img.shape[:2]\n\n    # Normalize to [0, 255]\n    img -= img.min()\n    if img.max() > 0:\n        img /= img.max()\n    img = (img * 255).astype(np.uint8)\n\n    # Resize\n    inter = cv2.INTER_AREA if (orig_h > size[0] or orig_w > size[1]) else cv2.INTER_LINEAR\n    img = cv2.resize(img, size, interpolation=inter)\n\n    # Save\n    cv2.imwrite(out_path, img)\n    return orig_h, orig_w\n\nprint('✅ dicom_to_png() function defined.')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 3.2  Test conversion on 1 sample ──────────────────────────────────────\n\n# ADD THIS LINE: Change the number 5 to any number (0 to 14999)\nsample_dcm_path = train_dicoms[3] \n\ntest_out = f'{OUTPUT_DIR}/sample_test.png'\nh, w = dicom_to_png(sample_dcm_path, test_out)\nprint(f'✅ Sample converted: original {h}×{w} → saved as 1024×1024 PNG')\n\nimg_check = cv2.imread(test_out, cv2.IMREAD_GRAYSCALE)\nprint(f'   Output shape   : {img_check.shape}')\nprint(f'   Pixel range    : [{img_check.min()}, {img_check.max()}]')\n\nplt.figure(figsize=(5, 5))\nplt.imshow(img_check, cmap='bone')\nplt.title(f'Sample PNG: {os.path.basename(sample_dcm_path)}', fontweight='bold')\nplt.axis('off')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## 📝 Section 4: Annotation Parsing & YOLO Format Conversion\n\nYOLO format per row:  \n```\n<class_id>  <x_center>  <y_center>  <width>  <height>\n```\nAll values are **normalized** to [0, 1] relative to image dimensions.","metadata":{}},{"cell_type":"code","source":"# ── 4.1  Define 14 class labels ────────────────────────────────────────────\nCLASS_NAMES = [\n    'Aortic enlargement',   # 0\n    'Atelectasis',           # 1\n    'Calcification',         # 2\n    'Cardiomegaly',          # 3\n    'Consolidation',         # 4\n    'ILD',                   # 5\n    'Infiltration',          # 6\n    'Lung Opacity',          # 7\n    'Nodule/Mass',           # 8\n    'Other lesion',          # 9\n    'Pleural effusion',      # 10\n    'Pleural thickening',    # 11\n    'Pneumothorax',          # 12\n    'Pulmonary fibrosis',    # 13\n    # 'No finding' is EXCLUDED from YOLO labels (no bounding box)\n]\n\nCLASS_TO_ID = {name: i for i, name in enumerate(CLASS_NAMES)}\n\nprint('Class → ID mapping:')\nfor name, cid in CLASS_TO_ID.items():\n    print(f'  [{cid:2d}] {name}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 4.2  Pre-build annotation lookup (group by image_id) ──────────────────\n# Filter out 'No finding' rows (they have no bbox)\ndf_findings = df[df['class_name'] != 'No finding'].copy()\ndf_findings['class_id'] = df_findings['class_name'].map(CLASS_TO_ID)\n\n# Drop rows where class_id is NaN (unknown classes)\ndf_findings.dropna(subset=['class_id'], inplace=True)\ndf_findings['class_id'] = df_findings['class_id'].astype(int)\n\n# Group by image_id for fast lookup\n# VinDr-CXR has multiple radiologists (rad_id) annotating the same image.\n# Strategy: Keep all annotations (consensus bboxes reduce noise naturally during training).\nannot_groups = df_findings.groupby('image_id')\n\nprint(f'Total images with findings : {len(annot_groups):,}')\nprint(f'Total finding annotations  : {len(df_findings):,}')\n\n# Images with no finding at all\nno_finding_ids = set(df[df['class_name'] == 'No finding']['image_id'])\nprint(f'Images labeled No Finding  : {len(no_finding_ids):,}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 4.3  CSV → YOLO conversion function ──────────────────────────────────\ndef csv_to_yolo(rows: pd.DataFrame, orig_h: int, orig_w: int) -> list:\n    \"\"\"\n    Convert annotation rows for a single image to YOLO format.\n\n    Args:\n        rows     : DataFrame rows for this image_id\n        orig_h   : original DICOM image height (pixels)\n        orig_w   : original DICOM image width  (pixels)\n\n    Returns:\n        List of strings: 'class_id x_c y_c w h' (all normalized 0-1)\n    \"\"\"\n    lines = []\n    for _, row in rows.iterrows():\n        x_min = float(row['x_min'])\n        y_min = float(row['y_min'])\n        x_max = float(row['x_max'])\n        y_max = float(row['y_max'])\n\n        # Skip degenerate boxes\n        if x_max <= x_min or y_max <= y_min:\n            continue\n\n        x_c = ((x_min + x_max) / 2) / orig_w\n        y_c = ((y_min + y_max) / 2) / orig_h\n        bw  = (x_max - x_min) / orig_w\n        bh  = (y_max - y_min) / orig_h\n\n        # Clip to [0, 1]\n        x_c = np.clip(x_c, 0.0, 1.0)\n        y_c = np.clip(y_c, 0.0, 1.0)\n        bw  = np.clip(bw,  0.0, 1.0)\n        bh  = np.clip(bh,  0.0, 1.0)\n\n        lines.append(f\"{int(row['class_id'])} {x_c:.6f} {y_c:.6f} {bw:.6f} {bh:.6f}\")\n    return lines\n\nprint('✅ csv_to_yolo() function defined.')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## ✂️ Section 5: Train / Validation Split & Full Processing Pipeline","metadata":{}},{"cell_type":"code","source":"# ── 5.1  Get all unique image IDs and split 80/20 ─────────────────────────\nall_image_ids = df['image_id'].unique().tolist()\ntrain_ids, val_ids = train_test_split(all_image_ids, test_size=0.20, random_state=42)\n\nprint(f'Total images    : {len(all_image_ids):,}')\nprint(f'Train set       : {len(train_ids):,} ({len(train_ids)/len(all_image_ids)*100:.1f}%)')\nprint(f'Validation set  : {len(val_ids):,}  ({len(val_ids)/len(all_image_ids)*100:.1f}%)')\n\n# Build a mapping: image_id → 'train' or 'val'\nsplit_map = {img_id: 'train' for img_id in train_ids}\nsplit_map.update({img_id: 'val' for img_id in val_ids})","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 5.2  Full pipeline: convert all DICOMs + write YOLO labels ────────────\n# Store original dims for YOLO coordinate scaling\norig_dims = {}   # { image_id: (orig_h, orig_w) }\nsuccess, failed = 0, 0\nerrors = []\n\nfor dcm_path in tqdm(train_dicoms, desc='Converting DICOM → PNG + YOLO labels'):\n    image_id = os.path.splitext(os.path.basename(dcm_path))[0]\n    split    = split_map.get(image_id, 'train')\n\n    img_out_dir = IMG_TRAIN if split == 'train' else IMG_VAL\n    lbl_out_dir = LBL_TRAIN if split == 'train' else LBL_VAL\n\n    png_path = os.path.join(img_out_dir, f'{image_id}.png')\n    lbl_path = os.path.join(lbl_out_dir, f'{image_id}.txt')\n\n    try:\n        # Convert DICOM → PNG and store original dimensions\n        orig_h, orig_w = dicom_to_png(dcm_path, png_path)\n        orig_dims[image_id] = (orig_h, orig_w)\n\n        # Write YOLO label file\n        if image_id in annot_groups.groups:\n            rows = annot_groups.get_group(image_id)\n            yolo_lines = csv_to_yolo(rows, orig_h, orig_w)\n        else:\n            yolo_lines = []  # 'No finding' image → empty label file\n\n        with open(lbl_path, 'w') as f:\n            f.write('\\n'.join(yolo_lines))\n\n        success += 1\n\n    except Exception as e:\n        failed += 1\n        errors.append((image_id, str(e)))\n\nprint(f'\\n✅ Conversion complete!')\nprint(f'   Successful : {success:,}')\nprint(f'   Failed     : {failed:,}')\nif errors:\n    print('\\nFailed files:')\n    for img_id, err in errors[:10]:\n        print(f'  {img_id}: {err}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 5.3  Verify output file counts ────────────────────────────────────────\nimg_train_count = len(glob.glob(os.path.join(IMG_TRAIN, '*.png')))\nimg_val_count   = len(glob.glob(os.path.join(IMG_VAL, '*.png')))\nlbl_train_count = len(glob.glob(os.path.join(LBL_TRAIN, '*.txt')))\nlbl_val_count   = len(glob.glob(os.path.join(LBL_VAL, '*.txt')))\n\nprint('='*40)\nprint('📂 OUTPUT DIRECTORY COUNTS')\nprint('='*40)\nprint(f'  images/train  : {img_train_count:,} PNGs')\nprint(f'  images/val    : {img_val_count:,} PNGs')\nprint(f'  labels/train  : {lbl_train_count:,} TXTs')\nprint(f'  labels/val    : {lbl_val_count:,} TXTs')\nprint()\n\n# Sanity check: image count == label count\nassert img_train_count == lbl_train_count, '❌ Mismatch in train set!'\nassert img_val_count   == lbl_val_count,   '❌ Mismatch in val set!'\nprint('✅ Image and label counts match for both splits!')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 5.4  Validate YOLO label format ───────────────────────────────────────\ndef validate_yolo_labels(label_dir: str, n_samples: int = 200) -> dict:\n    \"\"\"\n    Check that all values in YOLO labels are normalized within [0, 1].\n    Returns a dict with pass/fail counts.\n    \"\"\"\n    label_files = glob.glob(os.path.join(label_dir, '*.txt'))\n    sample_files = random.sample(label_files, min(n_samples, len(label_files)))\n\n    passed, failed_files = 0, []\n    for lbl_path in sample_files:\n        with open(lbl_path) as f:\n            lines = f.read().strip().splitlines()\n        ok = True\n        for line in lines:\n            parts = line.split()\n            if len(parts) != 5:\n                ok = False; break\n            vals = list(map(float, parts[1:]))\n            if not all(0.0 <= v <= 1.0 for v in vals):\n                ok = False; break\n        if ok:\n            passed += 1\n        else:\n            failed_files.append(lbl_path)\n    return {'checked': len(sample_files), 'passed': passed, 'failed': len(failed_files)}\n\nresult = validate_yolo_labels(LBL_TRAIN, n_samples=300)\nprint(f'Label validation (train): {result}')\n\nresult_val = validate_yolo_labels(LBL_VAL, n_samples=100)\nprint(f'Label validation (val)  : {result_val}')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## 🗂️ Section 6: Generate data.yaml for YOLOv8","metadata":{}},{"cell_type":"code","source":"# ── 6.1  Write data.yaml ───────────────────────────────────────────────────\nimport yaml\n\ndata_yaml = {\n    'path' : os.path.join(OUTPUT_DIR, 'dataset'),\n    'train': 'images/train',\n    'val'  : 'images/val',\n    'nc'   : len(CLASS_NAMES),\n    'names': CLASS_NAMES\n}\n\nyaml_path = os.path.join(OUTPUT_DIR, 'data.yaml')\nwith open(yaml_path, 'w') as f:\n    yaml.dump(data_yaml, f, default_flow_style=False, allow_unicode=True, sort_keys=False)\n\nprint('✅ data.yaml created at:', yaml_path)\nprint()\nwith open(yaml_path) as f:\n    print(f.read())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## 🖼️ Section 7: Visualization & Verification\nDisplay converted PNGs with their bounding boxes drawn on top.","metadata":{}},{"cell_type":"code","source":"# ── 7.1  Draw bounding boxes from YOLO label ───────────────────────────────\n# Colormap for 14 classes\nCOLORS = plt.cm.tab20(np.linspace(0, 1, 14))\n\ndef draw_yolo_boxes(png_path: str, lbl_path: str, ax, title: str = ''):\n    \"\"\"\n    Read a PNG and its YOLO label, draw bounding boxes on the given Axes.\n    \"\"\"\n    img = cv2.imread(png_path, cv2.IMREAD_GRAYSCALE)\n    H, W = img.shape\n    ax.imshow(img, cmap='bone')\n\n    with open(lbl_path) as f:\n        lines = f.read().strip().splitlines()\n\n    for line in lines:\n        if not line:\n            continue\n        parts = line.split()\n        cls  = int(parts[0])\n        x_c  = float(parts[1]) * W\n        y_c  = float(parts[2]) * H\n        bw   = float(parts[3]) * W\n        bh   = float(parts[4]) * H\n\n        x0 = x_c - bw / 2\n        y0 = y_c - bh / 2\n        color = COLORS[cls % 14]\n\n        rect = patches.Rectangle((x0, y0), bw, bh,\n                                  linewidth=2, edgecolor=color, facecolor='none')\n        ax.add_patch(rect)\n\n        label = CLASS_NAMES[cls] if cls < len(CLASS_NAMES) else str(cls)\n        ax.text(x0, y0 - 5, label, color=color, fontsize=7,\n                bbox=dict(boxstyle='round,pad=0.2', facecolor='black', alpha=0.5))\n\n    ax.set_title(title, fontsize=9, fontweight='bold')\n    ax.axis('off')\n\nprint('✅ draw_yolo_boxes() defined.')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 7.2  Display 8 sample images with bounding boxes ──────────────────────\npng_files = sorted(glob.glob(os.path.join(IMG_TRAIN, '*.png')))\n\n# Pick images that actually have annotations\nsamples = []\nfor p in png_files:\n    img_id = os.path.splitext(os.path.basename(p))[0]\n    lbl_p  = os.path.join(LBL_TRAIN, f'{img_id}.txt')\n    if os.path.exists(lbl_p) and os.path.getsize(lbl_p) > 0:\n        samples.append((p, lbl_p, img_id))\n    if len(samples) == 8:\n        break\n\nfig, axes = plt.subplots(2, 4, figsize=(20, 10))\nfig.suptitle('CliniScan – Sample Chest X-rays with YOLO Bounding Boxes\\n(Milestone 1 Verification)',\n             fontsize=15, fontweight='bold')\n\nfor i, (png_path, lbl_path, img_id) in enumerate(samples):\n    ax = axes[i // 4][i % 4]\n    draw_yolo_boxes(png_path, lbl_path, ax, title=img_id[:20])\n\nplt.tight_layout()\nplt.savefig(f'{OUTPUT_DIR}/milestone1_verification.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint('✅ Verification plot saved.')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 7.3  Class distribution after split ───────────────────────────────────\ndef count_classes_in_split(label_dir: str) -> dict:\n    counts = {n: 0 for n in CLASS_NAMES}\n    for lbl_path in glob.glob(os.path.join(label_dir, '*.txt')):\n        with open(lbl_path) as f:\n            for line in f:\n                line = line.strip()\n                if line:\n                    cls = int(line.split()[0])\n                    if cls < len(CLASS_NAMES):\n                        counts[CLASS_NAMES[cls]] += 1\n    return counts\n\ntrain_counts = count_classes_in_split(LBL_TRAIN)\nval_counts   = count_classes_in_split(LBL_VAL)\n\ncount_df = pd.DataFrame({'Train': train_counts, 'Val': val_counts})\n\nfig, ax = plt.subplots(figsize=(12, 6))\ncount_df.plot(kind='barh', ax=ax, color=['#2196F3', '#FF9800'])\nax.set_title('Class Distribution After Train/Val Split', fontsize=14, fontweight='bold')\nax.set_xlabel('Annotation Count')\nax.invert_yaxis()\nplt.tight_layout()\nplt.savefig(f'{OUTPUT_DIR}/class_distribution_split.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint(count_df.to_string())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## ✅ Section 8: Milestone 1 Summary Report","metadata":{}},{"cell_type":"code","source":"# ── 8.1  Final summary ─────────────────────────────────────────────────────\nprint('=' * 65)\nprint('  🫁  CliniScan – MILESTONE 1 COMPLETION REPORT')\nprint('=' * 65)\n\nprint(f'\\n📦  Dataset               : VinDr-CXR (Kaggle Competition)')\nprint(f'📊  Total Annotations     : {len(df):,} rows in train.csv')\nprint(f'🖼️   Total Images Converted: {success:,} DICOM → PNG (1024×1024)')\nprint(f'⚠️   Failed Conversions    : {failed}')\nprint()\nprint(f'✂️   Train/Val Split (80/20):')\nprint(f'    images/train  : {img_train_count:,}')\nprint(f'    images/val    : {img_val_count:,}')\nprint(f'    labels/train  : {lbl_train_count:,}')\nprint(f'    labels/val    : {lbl_val_count:,}')\nprint()\nprint(f'📁  Output Directory      : {OUTPUT_DIR}')\nprint(f'📄  data.yaml             : {yaml_path}')\nprint()\nprint('✅  Milestone 1 Objectives Completed:')\nprint('   [✓] Development environment set up')\nprint('   [✓] Dataset downloaded and explored (EDA)')\nprint('   [✓] DICOM → PNG conversion with VOI LUT windowing')\nprint('   [✓] Annotations parsed (CSV → YOLO format)')\nprint('   [✓] Train / Validation split (80/20)')\nprint('   [✓] data.yaml generated for YOLOv8')\nprint('   [✓] Sample visualization with bounding boxes')\nprint('   [✓] YOLO label validation passed')\nprint()\nprint('🚀  Ready for Milestone 2: Model Development & Baseline Training!')\nprint('=' * 65)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── 8.2  List all output files ─────────────────────────────────────────────\nprint('📂 Generated Artifacts:')\nfor root, dirs, files in os.walk(OUTPUT_DIR):\n    level = root.replace(OUTPUT_DIR, '').count(os.sep)\n    indent = '  ' * level\n    print(f'{indent}{os.path.basename(root)}/')\n    if level < 3:  # only show files at depth ≤ 2\n        subindent = '  ' * (level + 1)\n        for f in files:\n            size = os.path.getsize(os.path.join(root, f))\n            print(f'{subindent}{f}  ({size:,} bytes)')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Section 9: Classification Development (EfficientNet-B0) ──────────────────\nimport torch\nimport torch.nn as nn\nfrom torchvision import models, transforms\nfrom torch.utils.data import DataLoader, Dataset\n\nprint('🚀 Initializing Classification: EfficientNet-B0')\n\nclass CliniScanClassifyDataset(Dataset):\n    def __init__(self, csv_file, img_dir, transform=None):\n        self.data = pd.read_csv(csv_file)\n        self.img_dir = img_dir\n        self.transform = transform\n        # Binary target: No finding vs Abnormality\n        self.data['target'] = (self.data['class_name'] != 'No finding').astype(int)\n        self.images = self.data['image_id'].unique()[:1000] # Baseline subset\n\n    def __len__(self):\n        return len(self.images)\n\n    def __getitem__(self, idx):\n        img_id = self.images[idx]\n        img_path = os.path.join(self.img_dir, f'{img_id}.png')\n        if not os.path.exists(img_path):\n             return torch.zeros((3, 224, 224)), torch.tensor(0)\n             \n        image = cv2.imread(img_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        label = self.data[self.data['image_id'] == img_id]['target'].max()\n        \n        if self.transform:\n            image = self.transform(image)\n        return image, torch.tensor(label, dtype=torch.long)\n\n# Initialize Model\nmodel_ft = models.efficientnet_b0(weights='IMAGENET1K_V1')\nnum_ftrs = model_ft.classifier[1].in_features\nmodel_ft.classifier[1] = nn.Linear(num_ftrs, 2)\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel_ft = model_ft.to(device)\n\nprint(f'✅ Classification model ready on {device}.')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Section 10: Detection Development (YOLOv8s) ─────────────────────────────\n!pip install -q ultralytics\nfrom ultralytics import YOLO\n\nprint('🚀 Initializing Detection: YOLOv8s')\nmodel = YOLO('yolov8s.pt')\n\nprint(f'✅ YOLOv8s model loaded. Ready for training.')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This will run for 50 epochs using the data.yaml from Milestone 1\n# Ensure your GPU is turned ON in Kaggle settings!\nmodel.train(data=yaml_path, epochs=50, imgsz=1024, batch=16, device=0)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}