{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q pydicom opencv-python-headless ultralytics pandas numpy matplotlib scikit-learn\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport cv2\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport torch\nimport shutil\nimport random\n\n# What is CUDA? \n# CUDA is a computing platform by NVIDIA that allows PyTorch to use the GPU for parallel processing.\n# Using a GPU makes model training take minutes instead of days.\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\nif torch.cuda.is_available():\n    print(f\"GPU Name: {torch.cuda.get_device_name(0)}\")\nelse:\n    print(\"WARNING: No GPU found. Please turn on 'GPU T4 x2' or 'P100' in Kaggle settings.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Kaggle datasets can be nested inside /kaggle/input/ or /kaggle/input/competitions/.\n# This script recursively searches for the exact directory containing the 'train.csv' file.\nBASE_DIR = None\nfor root, dirs, files in os.walk('/kaggle/input'):\n    if 'train.csv' in files:\n        BASE_DIR = root\n        break\n\nif BASE_DIR is None:\n    raise FileNotFoundError(\"VinBigData dataset (train.csv) not found! Make sure you added the dataset to your notebook.\")\n\nTRAIN_DIR = os.path.join(BASE_DIR, 'train')\nCSV_PATH = os.path.join(BASE_DIR, 'train.csv')\n\nprint(f\"Selected BASE_DIR: {BASE_DIR}\")\ntrain_dicoms = sorted(glob.glob(os.path.join(TRAIN_DIR, '*.dicom')))\nprint(f\"Train DICOM files found: {len(train_dicoms):,}\")\n\nOUTPUT_DIR = '/kaggle/working/cliniscan'\nIMG_TRAIN = os.path.join(OUTPUT_DIR, 'dataset/images/train')\nLBL_TRAIN = os.path.join(OUTPUT_DIR, 'dataset/labels/train')\nIMG_VAL   = os.path.join(OUTPUT_DIR, 'dataset/images/val')\nLBL_VAL   = os.path.join(OUTPUT_DIR, 'dataset/labels/val')\n\nfor d in [IMG_TRAIN, IMG_VAL, LBL_TRAIN, LBL_VAL]:\n    os.makedirs(d, exist_ok=True)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(CSV_PATH)\ndf.fillna(0, inplace=True) \n\nclass_dict = {\n    0: 'Aortic enlargement', 1: 'Atelectasis', 2: 'Calcification', 3: 'Cardiomegaly',\n    4: 'Consolidation', 5: 'ILD', 6: 'Infiltration', 7: 'Lung Opacity',\n    8: 'Nodule/Mass', 9: 'Other lesion', 10: 'Pleural effusion', 11: 'Pleural thickening',\n    12: 'Pneumothorax', 13: 'Pulmonary fibrosis', 14: 'No finding'\n}\n\n# Generate consistent random colors for each class for visualization\nnp.random.seed(42)\nclass_colors = {i: [int(c) for c in np.random.randint(0, 255, size=3)] for i in range(15)}\n\n# Use ALL 15,000 unique images to satisfy Milestone 1 \"significant portion\"\nunique_images = df['image_id'].unique()\nnp.random.seed(42)\nsubset_images = unique_images  \n\n# 80/20 train/val split for the 15,000 images\nsplit_index = int(0.8 * len(subset_images))\ntrain_samples = set(subset_images[:split_index])\nval_samples = set(subset_images[split_index:])\n\ndf_subset = df[df['image_id'].isin(subset_images)]\nIMG_SIZE = 512\n\ndef process_dicom_to_png_and_yolo(img_id, split_name):\n    dicom_path = os.path.join(TRAIN_DIR, f\"{img_id}.dicom\")\n    if not os.path.exists(dicom_path): return False\n        \n    # pydicom.dcmread is the modern replacement for pydicom.read_file\n    dicom = pydicom.dcmread(dicom_path)\n    data = apply_voi_lut(dicom.pixel_array, dicom)\n    if dicom.PhotometricInterpretation == 'MONOCHROME1':\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    \n    original_h, original_w = data.shape\n    resized_img = cv2.resize(data, (IMG_SIZE, IMG_SIZE))\n    \n    img_folder = IMG_TRAIN if split_name == 'train' else IMG_VAL\n    cv2.imwrite(os.path.join(img_folder, f\"{img_id}.png\"), resized_img)\n    \n    label_folder = LBL_TRAIN if split_name == 'train' else LBL_VAL\n    img_annotations = df[df['image_id'] == img_id]\n    \n    yolo_lines = []\n    for _, row in img_annotations.iterrows():\n        class_id = int(row['class_id'])\n        if class_id == 14: continue # Skip \"No finding\" for detection boxes\n            \n        x_min, y_min = row['x_min'] / original_w, row['y_min'] / original_h\n        x_max, y_max = row['x_max'] / original_w, row['y_max'] / original_h\n        \n        x_center, y_center = (x_min + x_max) / 2, (y_min + y_max) / 2\n        width, height = x_max - x_min, y_max - y_min\n        yolo_lines.append(f\"{class_id} {x_center:.6f} {y_center:.6f} {width:.6f} {height:.6f}\")\n        \n    with open(os.path.join(label_folder, f\"{img_id}.txt\"), \"w\") as f:\n        f.write(\"\\n\".join(yolo_lines))\n    return True\n\nprint(f\"Converting all {len(subset_images)} DICOMs to PNG and generating YOLO labels...\")\nprint(\"WARNING: This full conversion will take approx. 2.5 hours on Kaggle.\")\nfor img_id in tqdm(subset_images):\n    split = 'train' if img_id in train_samples else 'val'\n    process_dicom_to_png_and_yolo(img_id, split)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_sample(img_id, split='train'):\n    folder_img = IMG_TRAIN if split == 'train' else IMG_VAL\n    folder_lbl = LBL_TRAIN if split == 'train' else LBL_VAL\n    \n    img_path = os.path.join(folder_img, f\"{img_id}.png\")\n    lbl_path = os.path.join(folder_lbl, f\"{img_id}.txt\")\n    \n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    if os.path.exists(lbl_path):\n        with open(lbl_path, \"r\") as f:\n            lines = f.readlines()\n            for line in lines:\n                parts = line.strip().split()\n                if len(parts) == 5:\n                    c_id, xc, yc, w, h = int(parts[0]), float(parts[1]), float(parts[2]), float(parts[3]), float(parts[4])\n                    x_min, y_min = int((xc - w/2) * IMG_SIZE), int((yc - h/2) * IMG_SIZE)\n                    x_max, y_max = int((xc + w/2) * IMG_SIZE), int((yc + h/2) * IMG_SIZE)\n                    \n                    color = class_colors[c_id]\n                    label = class_dict[c_id]\n                    \n                    # Draw bounding box\n                    cv2.rectangle(img, (x_min, y_min), (x_max, y_max), color, 2)\n                    \n                    # Add background block for text to make it readable\n                    (tw, th), _ = cv2.getTextSize(label, cv2.FONT_HERSHEY_SIMPLEX, 0.4, 1)\n                    cv2.rectangle(img, (x_min, y_min - th - 4), (x_min + tw, y_min), color, -1)\n                    \n                    # Add text\n                    cv2.putText(img, label, (x_min, y_min - 2), cv2.FONT_HERSHEY_SIMPLEX, 0.4, (255, 255, 255), 1, cv2.LINE_AA)\n    return img\n\nimages_with_findings = df_subset[df_subset['class_id'] != 14]['image_id'].unique()\nsample_plot_ids = images_with_findings[:8]\n\nplt.figure(figsize=(24, 12))\nplt.suptitle(\"CliniScan - Sample Chest X-rays with YOLO Bounding Boxes\\n(Milestone 1 Verification)\", fontsize=20, weight='bold')\nfor i, img_id in enumerate(sample_plot_ids):\n    split = 'train' if img_id in train_samples else 'val'\n    plt.subplot(2, 4, i+1)\n    plt.imshow(visualize_sample(img_id, split))\n    plt.title(img_id, fontsize=10)\n    plt.axis('off')\nplt.tight_layout(rect=[0, 0.03, 1, 0.95])\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n--- Starting Classification Baseline Training (Abnormal vs Normal) ---\")\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models, transforms\nimport torch.nn as nn\nimport torch.optim as optim\nfrom sklearn.metrics import roc_auc_score, f1_score\n\nclass CliniScanBinaryDataset(Dataset):\n    def __init__(self, df_subset, img_ids, base_img_dir):\n        self.img_ids = list(img_ids)\n        self.labels = {}\n        for img_id in self.img_ids:\n            img_rows = df_subset[df_subset['image_id'] == img_id]\n            is_normal = all(img_rows['class_name'] == 'No finding')\n            self.labels[img_id] = 0 if is_normal else 1\n            \n        self.transform = transforms.Compose([\n            transforms.ToTensor(),\n            transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n        ])\n\n    def __len__(self): return len(self.img_ids)\n        \n    def __getitem__(self, idx):\n        img_id = self.img_ids[idx]\n        split = 'train' if img_id in train_samples else 'val'\n        folder = IMG_TRAIN if split == 'train' else IMG_VAL\n        \n        img = cv2.imread(os.path.join(folder, f\"{img_id}.png\"))\n        img = cv2.resize(img, (224, 224))\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        if self.transform: img = self.transform(img)\n        return img, torch.tensor(self.labels[img_id], dtype=torch.float32)\n\n# Note: The batch size is reduced to 8 to fit the big dataset on Kaggle GPUs\ntrain_loader = DataLoader(CliniScanBinaryDataset(df_subset, train_samples, OUTPUT_DIR), batch_size=8, shuffle=True)\nval_loader = DataLoader(CliniScanBinaryDataset(df_subset, val_samples, OUTPUT_DIR), batch_size=8, shuffle=False)\n\nmodel_cls = models.efficientnet_b0(weights=models.EfficientNet_B0_Weights.DEFAULT)\nmodel_cls.classifier[1] = nn.Linear(model_cls.classifier[1].in_features, 1) # Binary output\nmodel_cls = model_cls.to(device)\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model_cls.parameters(), lr=0.001)\n\nepochs = 2 \nfor epoch in range(epochs):\n    model_cls.train()\n    total_loss = 0\n    for images, labels in train_loader:\n        images, labels = images.to(device), labels.to(device).unsqueeze(1)\n        optimizer.zero_grad()\n        loss = criterion(model_cls(images), labels)\n        loss.backward()\n        optimizer.step()\n        total_loss += loss.item()\n    print(f\"Classification Epoch {epoch+1}/{epochs} - Train Loss: {total_loss/len(train_loader):.4f}\")\n\n# Classification Evaluation Metrics\nmodel_cls.eval()\nall_preds, all_labels = [], []\nwith torch.no_grad():\n    for images, labels in val_loader:\n        all_preds.extend(torch.sigmoid(model_cls(images.to(device))).cpu().numpy())\n        all_labels.extend(labels.numpy())\n\npred_binary = (np.array(all_preds) > 0.5).astype(int)\nauc = roc_auc_score(all_labels, all_preds)\nf1 = f1_score(all_labels, pred_binary)\nprint(f\"\\n✅ Baseline Classification Metrics -> AUC: {auc:.4f} | F1-Score: {f1:.4f}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n--- Starting Detection Baseline Training (YOLOv8) ---\")\nfrom ultralytics import YOLO\n\nyaml_content = f\"\"\"\npath: {OUTPUT_DIR}/dataset\ntrain: images/train\nval: images/val\nnc: 14\nnames: ['Aortic enlargement', 'Atelectasis', 'Calcification', 'Cardiomegaly', 'Consolidation', 'ILD', 'Infiltration', 'Lung Opacity', 'Nodule/Mass', 'Other lesion', 'Pleural effusion', 'Pleural thickening', 'Pneumothorax', 'Pulmonary fibrosis']\n\"\"\"\nwith open(os.path.join(OUTPUT_DIR, \"data.yaml\"), \"w\") as f:\n    f.write(yaml_content)\n\nmodel_det = YOLO('yolov8s.pt')\n# Train YOLO for 3 epochs on the full dataset\nmodel_det.train(data=os.path.join(OUTPUT_DIR, \"data.yaml\"), epochs=3, imgsz=IMG_SIZE, project=OUTPUT_DIR, name=\"yolo_baseline\", batch=8)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import Image, display\n\nresults_path = os.path.join(OUTPUT_DIR, \"yolo_baseline\", \"results.png\")\nif os.path.exists(results_path):\n    print(\"\\n--- YOLOv8 Baseline Metrics Chart ---\")\n    display(Image(filename=results_path, width=800))\n\nval_preds = glob.glob(os.path.join(OUTPUT_DIR, \"yolo_baseline\", \"val_batch*_pred.jpg\"))\nif val_preds:\n    print(\"\\n--- YOLOv8 Baseline Validation Predictions ---\")\n    display(Image(filename=val_preds[0], width=800))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}