{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":11215393,"datasetId":6799157,"databundleVersionId":11623300},{"sourceType":"datasetVersion","sourceId":11339939,"datasetId":7094420,"databundleVersionId":11766384},{"sourceType":"datasetVersion","sourceId":2057341,"datasetId":1232864,"databundleVersionId":2097467},{"sourceType":"datasetVersion","sourceId":8785422,"datasetId":5281464,"databundleVersionId":8941916},{"sourceType":"datasetVersion","sourceId":1799839,"datasetId":1069682,"databundleVersionId":1837296},{"sourceType":"datasetVersion","sourceId":1799615,"datasetId":1069544,"databundleVersionId":1837072},{"sourceType":"modelInstanceVersion","sourceId":309588,"databundleVersionId":11623012,"modelInstanceId":262719}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. ***Libraries Installation & Importing*** ","metadata":{}},{"cell_type":"code","source":"!pip install ultralytics\n!pip install pydicom Pillow\n!pip install scikit-image\n!pip install tqdm --upgrade\n!pip install scikit-learn\n!pip install -q ensemble-boxes","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:33:46.254402Z","iopub.execute_input":"2025-05-12T07:33:46.254684Z","iopub.status.idle":"2025-05-12T07:34:09.263226Z","shell.execute_reply.started":"2025-05-12T07:33:46.254653Z","shell.execute_reply":"2025-05-12T07:34:09.262149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────\n# ✅ Standard libraries\n# ─────────────────────────────\nimport os\nimport gc\nimport shutil\nimport random\nimport warnings\nfrom glob import glob\n\n# ─────────────────────────────\n# ✅ Data handling\n# ─────────────────────────────\nimport numpy as np\nimport pandas as pd\nimport yaml\nfrom tqdm.autonotebook import tqdm\n\n# ─────────────────────────────\n# ✅ Image handling\n# ─────────────────────────────\nimport cv2\nimport albumentations as A\nimport skimage.io\nimport skimage.transform\n\n# ─────────────────────────────\n# ✅ Machine learning & utilities\n# ─────────────────────────────\nfrom sklearn.model_selection import StratifiedGroupKFold\n\n# ─────────────────────────────\n# ✅ Deep learning\n# ─────────────────────────────\nimport torch\nimport torch.nn.functional as F\nimport torchvision\nfrom torch.utils.data import DataLoader, Dataset\n\n# ─────────────────────────────\n# ✅ Object Detection (YOLO & WBF)\n# ─────────────────────────────\nfrom ultralytics import YOLO\nfrom ensemble_boxes import weighted_boxes_fusion\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:34:09.264277Z","iopub.execute_input":"2025-05-12T07:34:09.264527Z","iopub.status.idle":"2025-05-12T07:34:19.218854Z","shell.execute_reply.started":"2025-05-12T07:34:09.264503Z","shell.execute_reply":"2025-05-12T07:34:19.217959Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Data Preparation, Bounding Box Visualization, and Class Distribution","metadata":{}},{"cell_type":"code","source":"# ✅ Load dataset\nlabel_data_file = \"/kaggle/input/vinbigdata-1024-image-dataset/vinbigdata/train.csv\"\ntrain_df = pd.read_csv(label_data_file)\n\n# ✅ Add image_path column\ntrain_df['image_path'] = '/kaggle/input/vinbigdata-1024-image-dataset/vinbigdata/train/' + train_df.image_id + '.png'\n\n# ✅ Remove class 14 (No Finding) and class 2 (Calcification) completely\ntrain_df = train_df[~train_df.class_id.isin([14, 2])].reset_index(drop=True)\n\n# ✅ Print remaining images to confirm\nprint(f\"✅ Number of images remaining: {train_df['image_id'].nunique()}\")\n\n# ✅ Convert VinBigData bbox format to YOLO format\ntrain_df['x_mid'] = (train_df['x_min'] + train_df['x_max']) / (2 * train_df['width'])\ntrain_df['y_mid'] = (train_df['y_min'] + train_df['y_max']) / (2 * train_df['height'])\ntrain_df['w'] = (train_df['x_max'] - train_df['x_min']) / train_df['width']\ntrain_df['h'] = (train_df['y_max'] - train_df['y_min']) / train_df['height']\n\ntrain_df['source_dataset'] = 'vinbig'\n\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:34:19.220502Z","iopub.execute_input":"2025-05-12T07:34:19.220968Z","iopub.status.idle":"2025-05-12T07:34:19.51148Z","shell.execute_reply.started":"2025-05-12T07:34:19.220942Z","shell.execute_reply":"2025-05-12T07:34:19.510702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Load the new NIH dataset\nnew_nih_file = \"/kaggle/input/nih-chest-xray-dataset-bbox-for-vinbigdata/nih.csv\"\nnew_nih_df = pd.read_csv(new_nih_file)\n\n# ✅ Add image path for new NIH dataset\nnew_nih_df['image_path'] = '/kaggle/input/nih-chest-xray-dataset-bbox-for-vinbigdata/nih/' + new_nih_df['image_id'] + '.png'\n\n# ✅ Remove rows with unmapped class names (NaN)\nnew_nih_df = new_nih_df.dropna(subset=['class_name'])\n\n# ✅ Define the class_name_to_id mapping\nclass_name_to_id = {\n    \"Aortic enlargement\": 0,\n    \"Cardiomegaly\": 2,  \n    \"Consolidation\": 3,\n    \"ILD\": 4,\n    \"Infiltration\": 5,\n    \"Lung Opacity\": 6,\n    \"Nodule/Mass\": 7,\n    \"Other lesion\": 8,\n    \"Pleural effusion\": 9,\n    \"Pleural thickening\": 10,\n    \"Pneumothorax\": 11,\n    \"Pulmonary fibrosis\": 12,\n    \"Atelectasis\": 1\n}\n\n# ✅ Assign class_id based on class_name\nnew_nih_df[\"class_id\"] = new_nih_df[\"class_name\"].map(class_name_to_id)\n\n# ✅ Calculate actual image width and height\nimage_widths = []\nimage_heights = []\n\nfor path in new_nih_df[\"image_path\"]:\n    image = cv2.imread(path)\n    if image is not None:\n        height, width = image.shape[:2]\n    else:\n        height, width = -1, -1  # Handle missing/corrupted image case\n    image_widths.append(width)\n    image_heights.append(height)\n\n# ✅ Store as 'width' and 'height' (actual image dimensions, not bbox)\nnew_nih_df[\"width\"] = image_widths\nnew_nih_df[\"height\"] = image_heights\n\n# ✅ Convert bbox to YOLO format (1-step calculation + normalization)\nnew_nih_df['x_mid'] = (new_nih_df['x_min'] + new_nih_df['x_max']) / (2 * new_nih_df['width'])\nnew_nih_df['y_mid'] = (new_nih_df['y_min'] + new_nih_df['y_max']) / (2 * new_nih_df['height'])\nnew_nih_df['w'] = (new_nih_df['x_max'] - new_nih_df['x_min']) / new_nih_df['width']\nnew_nih_df['h'] = (new_nih_df['y_max'] - new_nih_df['y_min']) / new_nih_df['height']\n\nnew_nih_df['source_dataset'] = 'nih_for_vin'\n\n# ✅ Select relevant columns (now width/height = image size)\nnew_nih_df = new_nih_df[[\n    'image_id', 'class_name', 'class_id', 'rad_id',\n    'x_mid', 'y_mid', 'w', 'h',\n    'x_min', 'y_min', 'x_max', 'y_max',\n    'width', 'height', 'image_path', 'source_dataset'\n]]\n\n# ✅ Filter for only the classes you want to add: Atelectasis, Pneumothorax, and Nodule/Mass\nrelevant_classes = [\"Atelectasis\", \"Pneumothorax\", \"Nodule/Mass\"]\nfiltered_nih_df = new_nih_df[new_nih_df[\"class_name\"].isin(relevant_classes)]\n\n# ✅ Merge the filtered DataFrame with the existing train_df\ntrain_df = pd.concat([train_df, filtered_nih_df], ignore_index=True)\n\n# ✅ Check for NaN class IDs after merging\nprint(f\"Number of NaN class IDs after merge: {train_df['class_id'].isna().sum()}\")  # Should be 0\n\n# ✅ Print final dataset details\nprint(f\"✅ Number of unique images after final merge: {train_df['image_id'].nunique()}\")\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:34:19.512777Z","iopub.execute_input":"2025-05-12T07:34:19.513071Z","iopub.status.idle":"2025-05-12T07:34:45.312158Z","shell.execute_reply.started":"2025-05-12T07:34:19.513047Z","shell.execute_reply":"2025-05-12T07:34:45.311311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Load NIH dataset\nnih_bbox_file = \"/kaggle/input/nih-chest-x-rays-bbox-version/BBox_List_2017.csv\"\nnih_df = pd.read_csv(nih_bbox_file)\n\n# ✅ Add image path\nnih_df['image_path'] = '/kaggle/input/nih-chest-x-rays-bbox-version/bbox_img/' + nih_df['Image Index']\n\n# ✅ Map class names\nnih_class_mapping = {\n    \"Infiltrate\": \"Infiltration\",\n    \"Atelectasis\": \"Atelectasis\",\n    \"Pneumonia\": \"Pneumonia\",\n    \"Cardiomegaly\": \"Cardiomegaly\",\n    \"Effusion\": \"Pleural effusion\",\n    \"Pneumothorax\": \"Pneumothorax\",\n    \"Mass\": \"Nodule/Mass\",\n    \"Nodule\": \"Nodule/Mass\"\n}\nnih_df['class_name'] = nih_df['Finding Label'].map(nih_class_mapping)\nnih_df = nih_df.dropna(subset=['class_name'])\n\n# ✅ Rename bbox columns\nnih_df = nih_df.rename(columns={\n    \"Image Index\": \"image_id\",\n    \"Bbox [x\": \"x_min\",\n    \"y\": \"y_min\",\n    \"w\": \"w\",\n    \"h]\": \"h\"\n})\n\n# ✅ Compute x_max, y_max\nnih_df['x_max'] = nih_df['x_min'] + nih_df['w']\nnih_df['y_max'] = nih_df['y_min'] + nih_df['h']\n\n# ✅ Assume fixed image size if actual dimensions are not available (e.g., 1024x1024)\nnih_df['width'] = 1024\nnih_df['height'] = 1024\n\n# ✅ Compute YOLO format in one step\nnih_df['x_mid'] = (nih_df['x_min'] + nih_df['x_max']) / (2 * nih_df['width'])\nnih_df['y_mid'] = (nih_df['y_min'] + nih_df['y_max']) / (2 * nih_df['height'])\nnih_df['w'] = (nih_df['x_max'] - nih_df['x_min']) / nih_df['width']\nnih_df['h'] = (nih_df['y_max'] - nih_df['y_min']) / nih_df['height']\n\nnih_df['source_dataset'] = 'nih'\n\n# ✅ Final column selection\nnih_df = nih_df[['image_id', 'class_name', 'x_mid', 'y_mid', 'w', 'h', 'x_min', 'y_min', 'x_max', 'y_max', 'width', 'height', 'image_path','source_dataset']]\n\n# ✅ Filter for only the classes you want to add: Atelectasis, Pneumothorax, and Nodule/Mass\nrelevant_classes = [\"Atelectasis\", \"Pneumothorax\", \"Nodule/Mass\"]\nfiltered_nih_df = nih_df[nih_df[\"class_name\"].isin(relevant_classes)]\n\n# ✅ Merge the filtered DataFrame with the existing train_df\ntrain_df = pd.concat([train_df, filtered_nih_df], ignore_index=True)\n\n# ✅ Assign class_id to the merged dataset\nclass_name_to_id = {\n    \"Aortic enlargement\": 0,\n    \"Cardiomegaly\": 2,  \n    \"Consolidation\": 3,\n    \"ILD\": 4,\n    \"Infiltration\": 5,\n    \"Lung Opacity\": 6,\n    \"Nodule/Mass\": 7,\n    \"Other lesion\": 8,\n    \"Pleural effusion\": 9,\n    \"Pleural thickening\": 10,\n    \"Pneumothorax\": 11,\n    \"Pulmonary fibrosis\": 12,\n    \"Atelectasis\": 1,\n    \"Pneumonia\": 13 \n}\ntrain_df[\"class_id\"] = train_df[\"class_name\"].map(class_name_to_id)\n\n# ✅ Check for NaN class IDs after merging\nprint(f\"Number of NaN class IDs: {train_df['class_id'].isna().sum()}\")  # Should be 0\n\n# ✅ Print final dataset details\nprint(f\"✅ Number of unique images after merging: {train_df['image_id'].nunique()}\")\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:34:45.313108Z","iopub.execute_input":"2025-05-12T07:34:45.31336Z","iopub.status.idle":"2025-05-12T07:34:45.37306Z","shell.execute_reply.started":"2025-05-12T07:34:45.31334Z","shell.execute_reply":"2025-05-12T07:34:45.372207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport pandas as pd\n\n# Define the directories where the images and their corresponding label files are stored\nimages_dir = \"/kaggle/input/chestxrayabnormalities/train/images\"  # Correct directory\nlabels_dir = \"/kaggle/input/chestxrayabnormalities/train/labels\"  # Correct directory\n\n# Class mapping (ID to class name) based on the ChestX-ray dataset\nclass_mapping = {\n    0: \"Aortic_enlargement\",\n    1: \"Atelectasis\",\n    2: \"Calcification\",\n    3: \"Cardiomegaly\",\n    4: \"Consolidation\",\n    5: \"ILD\",\n    6: \"Infiltration\",\n    7: \"Lung_Opacity\",\n    8: \"Nodule-Mass\",  # This is mapped to 'Nodule/Mass'\n    9: \"Other_lesion\",\n    10: \"Pleural_effusion\",\n    11: \"Pleural_thickening\",\n    12: \"Pneumothorax\",\n    13: \"Pulmonary_fibrosis\"\n}\n\n# Define the relevant classes you want to keep\nrelevant_classes = [\"Atelectasis\", \"Pneumothorax\", \"Nodule-Mass\"]\n\n# Create a list to hold the processed data\nprocessed_data = []\n\n# Loop through all the label files in the directory\nfor label_file in os.listdir(labels_dir):\n    if label_file.endswith('.txt'):\n        # Extract the image ID (without extension and remove _jpg.rf.* part from the filename)\n        image_id = label_file.split('.')[0]\n        \n        # Remove the unwanted parts like _jpg.rf.XXX from the image_id\n        image_id_cleaned = image_id.split('_jpg')[0]\n        \n        # Construct the image path correctly (append .jpg)\n        image_path = os.path.join(images_dir, f\"{image_id_cleaned}.jpg\")\n        \n        # Read the image to get its dimensions (height and width)\n        image = cv2.imread(image_path)\n        if image is not None:\n            image_height, image_width = image.shape[:2]\n        else:\n            image_height, image_width = -1, -1  # Handle missing or corrupted image case\n        \n        # Read the corresponding label file\n        label_file_path = os.path.join(labels_dir, label_file)\n        with open(label_file_path, 'r') as f:\n            lines = f.readlines()\n        \n        # Process each line in the label file\n        for line in lines:\n            parts = line.strip().split()\n            class_id = int(parts[0])  # Original class ID\n\n            # Only process the class_ids that are in the relevant_classes set\n            if class_mapping.get(class_id) in relevant_classes:\n                class_name = class_mapping[class_id]\n                \n                # YOLO format is already in normalized form\n                x_mid = float(parts[1])\n                y_mid = float(parts[2])\n                bbox_width = float(parts[3])\n                bbox_height = float(parts[4])\n\n                # Append the data for this image and label (including class_id and class_name)\n                processed_data.append([image_id, class_name, class_id, x_mid, y_mid, bbox_width, bbox_height])\n\n# Convert the processed data into a DataFrame\nprocessed_df = pd.DataFrame(processed_data, columns=['image_id', 'class_name', 'class_id', 'x_mid', 'y_mid', 'w', 'h'])\n\n# Construct the image path correctly (remove unwanted parts of image_id and append .jpg)\nprocessed_df['image_path'] = processed_df['image_id'].apply(lambda x: os.path.join(images_dir, f\"{x.split('_jpg')[0]}.jpg\"))\n\n# Add the source dataset name (in this case, 'chestxrayabnormalities')\nprocessed_df['source_dataset'] = 'chestxrayabnormalities'\n\n# Standardize class names to match the format in train_df (e.g., 'Nodule-Mass' to 'Nodule/Mass')\nprocessed_df['class_name'] = processed_df['class_name'].replace(\"Nodule-Mass\", \"Nodule/Mass\")\n\n# Define the class_name to class_id mapping for the final step\nclass_name_to_id = {\n    \"Aortic_enlargement\": 0,\n    \"Atelectasis\": 1,\n    \"Calcification\": 2,\n    \"Cardiomegaly\": 3,\n    \"Consolidation\": 4,\n    \"ILD\": 5,\n    \"Infiltration\": 6,\n    \"Lung_Opacity\": 7,\n    \"Nodule/Mass\": 8,  # Ensure it matches with 'Nodule/Mass' for consistency\n    \"Other_lesion\": 9,\n    \"Pleural_effusion\": 10,\n    \"Pleural_thickening\": 11,\n    \"Pneumothorax\": 12,\n    \"Pulmonary_fibrosis\": 13\n}\n\n# Assign class_id based on class_name\nprocessed_df['class_id'] = processed_df['class_name'].map(class_name_to_id)\n\n# Check for any NaN values in class_id (should be 0 if no issue)\nprint(f\"✅ Number of NaN class IDs: {processed_df['class_id'].isna().sum()}\")  # Should be 0\n\n# Print the first few rows of the final DataFrame\nprint(f\"✅ Final dataset preview:\")\nprint(processed_df.head())\n\n# ✅ Now, merge with train_df (if exists) or create a new train_df\ntrain_df = pd.concat([train_df, processed_df], ignore_index=True)\n\n# ✅ Print final dataset details\nprint(f\"✅ Number of unique images after merging: {train_df['image_id'].nunique()}\")\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:34:45.37401Z","iopub.execute_input":"2025-05-12T07:34:45.374264Z","iopub.status.idle":"2025-05-12T07:36:49.961769Z","shell.execute_reply.started":"2025-05-12T07:34:45.374243Z","shell.execute_reply":"2025-05-12T07:36:49.960902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Unique classes per image\nunique_class_per_image = train_df.groupby(\"image_id\")[\"class_name\"].unique()\n\n# Count how many images each class appears in\nclass_image_counts = unique_class_per_image.explode().value_counts().sort_values(ascending=False)\n\n# Convert to DataFrame with class_id\nclass_name_to_id = train_df.drop_duplicates(\"class_name\")[[\"class_name\", \"class_id\"]].set_index(\"class_name\")[\"class_id\"].to_dict()\n\n# Final DataFrame\nclass_image_stats = pd.DataFrame({\n    \"class_name\": class_image_counts.index,\n    \"class_id\": [class_name_to_id[c] for c in class_image_counts.index],\n    \"num_images\": class_image_counts.values\n}).reset_index(drop=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:36:49.962749Z","iopub.execute_input":"2025-05-12T07:36:49.963081Z","iopub.status.idle":"2025-05-12T07:36:50.216724Z","shell.execute_reply.started":"2025-05-12T07:36:49.96305Z","shell.execute_reply":"2025-05-12T07:36:50.21571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Total bounding boxes per class\nclass_counts = train_df['class_name'].value_counts().sort_values(ascending=False)\n\n# Map class_name to class_id\nclass_name_to_id = train_df.drop_duplicates(\"class_name\")[[\"class_name\", \"class_id\"]].set_index(\"class_name\")[\"class_id\"].to_dict()\n\n# Create final DataFrame\nclass_box_distribution = pd.DataFrame({\n    \"class_name\": class_counts.index,\n    \"class_id\": [class_name_to_id[cls] for cls in class_counts.index],\n    \"num_boxes\": class_counts.values\n}).reset_index(drop=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:36:50.217739Z","iopub.execute_input":"2025-05-12T07:36:50.218064Z","iopub.status.idle":"2025-05-12T07:36:50.228299Z","shell.execute_reply.started":"2025-05-12T07:36:50.218034Z","shell.execute_reply":"2025-05-12T07:36:50.227337Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Class Mapping","metadata":{}},{"cell_type":"code","source":"# ✅ Define the class mapping dictionary with old IDs (0 to 13) and their corresponding new groupings\nclass_mapping = {\n    0: \"Aortic Enlargement\",    # Aortic enlargement (ID 0)\n    1: \"Lung Collapse\",         # Atelectasis (ID 1)\n    2: \"Cardiomegaly\",          # Cardiomegaly (ID 2)\n    3: \"Opacities/Infiltration\",  # Consolidation (ID 3)\n    4: \"Opacities/Infiltration\",  # ILD (ID 4)\n    5: \"Opacities/Infiltration\",  # Infiltration (ID 5)\n    6: \"Opacities/Infiltration\",  # Lung Opacity (ID 6)\n    7: \"Nodule/Mass\",           # Nodule/Mass (ID 7)\n    8: \"Nodule/Mass\",           # Nodule/Mass (ID 8)\n    9: \"Pleural Conditions\",    # Pleural Effusion (ID 9)\n    10: \"Pleural Conditions\",   # Pleural Thickening (ID 10)\n    11: \"Lung Collapse\",        # Pneumothorax (ID 11)\n    12: \"Opacities/Infiltration\",  # Pulmonary fibrosis (ID 12)\n    13: \"Opacities/Infiltration\"   # Lung Opacity (ID 13)\n}\n\n# ✅ Apply the class mapping to create `mapped_class_name`\ntrain_df['mapped_class_name'] = train_df['class_id'].map(class_mapping)\n\n# ✅ Debugging: Check for unmapped class IDs\nunmapped_classes = train_df[train_df['mapped_class_name'].isna()]['class_id'].unique()\nif len(unmapped_classes) > 0:\n    print(f\"⚠️ Warning: Some class IDs are not mapped! Unmapped class IDs: {unmapped_classes}\")\n\n# ✅ Remove any NaN values before creating unique class names\ntrain_df = train_df.dropna(subset=['mapped_class_name'])\n\n# ✅ Get unique class names (ensuring correct count)\nclass_names = sorted(train_df['mapped_class_name'].unique())\n\n# ✅ Explicitly map class names to new indices\nnew_class_ids = {name: idx for idx, name in enumerate(class_names)}\n\n# ✅ Apply the new mapping to create a new class ID\ntrain_df['new_class_id'] = train_df['mapped_class_name'].map(new_class_ids)\n\n# ✅ Create the final new class mapping (new class IDs)\nnew_class_mapping = {idx: name for name, idx in new_class_ids.items()}\n\n# ✅ Verify the new class mapping\nprint(f\"✅ New class mapping (new class IDs): {new_class_mapping}\")\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:36:50.229232Z","iopub.execute_input":"2025-05-12T07:36:50.22952Z","iopub.status.idle":"2025-05-12T07:36:50.292457Z","shell.execute_reply.started":"2025-05-12T07:36:50.229492Z","shell.execute_reply":"2025-05-12T07:36:50.2916Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"WBF","metadata":{}},{"cell_type":"code","source":"from collections import Counter  # Add this import\n# ✅ Apply WBF Function\ndef apply_wbf(train_df, iou_thr=0.1, skip_box_thr=0.0001, min_box_size=0.01):\n    output = []\n\n    for image_id, group in tqdm(train_df.groupby(\"image_id\"), desc=\"Applying WBF\"):\n        w, h = group['width'].iloc[0], group['height'].iloc[0]\n\n        boxes_list = []\n        scores_list = []\n        labels_list = []\n\n        boxes_single = []\n        labels_single = []\n\n        count_dict = Counter(group['new_class_id'].tolist())\n        class_ids = group['new_class_id'].unique().tolist()\n\n        for cid in class_ids:\n            class_group = group[group.new_class_id == cid]\n\n            if count_dict[cid] == 1:\n                row = class_group.iloc[0]\n                # Use YOLO-normalized box and convert to (x_min, y_min, x_max, y_max)\n                x_mid, y_mid, box_w, box_h = row['x_mid'], row['y_mid'], row['w'], row['h']\n                x_min = x_mid - box_w / 2\n                y_min = y_mid - box_h / 2\n                x_max = x_mid + box_w / 2\n                y_max = y_mid + box_h / 2\n\n                box = [x_min, y_min, x_max, y_max]\n                boxes_single.append(box)\n                labels_single.append(cid)\n            else:\n                # Same as above, use YOLO-normalized coords\n                x_mid = class_group['x_mid'].to_numpy()\n                y_mid = class_group['y_mid'].to_numpy()\n                box_w = class_group['w'].to_numpy()\n                box_h = class_group['h'].to_numpy()\n\n                x_min = x_mid - box_w / 2\n                y_min = y_mid - box_h / 2\n                x_max = x_mid + box_w / 2\n                y_max = y_mid + box_h / 2\n\n                bboxes = np.stack([x_min, y_min, x_max, y_max], axis=1)\n                bboxes = np.clip(bboxes, 0, 1)  # Ensure in [0,1]\n\n                boxes_list.append(bboxes.tolist())\n                scores_list.append([1.0] * len(class_group))\n                labels_list.append([cid] * len(class_group))\n\n        # Apply WBF\n        if boxes_list:\n            fused_boxes, _, fused_labels = weighted_boxes_fusion(\n                boxes_list, scores_list, labels_list,\n                weights=None, iou_thr=iou_thr, skip_box_thr=skip_box_thr\n            )\n        else:\n            fused_boxes, fused_labels = np.empty((0, 4)), np.empty((0,))\n\n        # Combine with singles\n        if len(boxes_single) > 0:\n            all_boxes = np.vstack([fused_boxes, boxes_single])\n            all_labels = np.hstack([fused_labels, labels_single])\n        else:\n            all_boxes = fused_boxes\n            all_labels = fused_labels\n\n        # Convert back to YOLO format and append\n        for box, label in zip(all_boxes, all_labels):\n            x_min, y_min, x_max, y_max = box\n            box_w = x_max - x_min\n            box_h = y_max - y_min\n            x_center = (x_min + x_max) / 2\n            y_center = (y_min + y_max) / 2\n\n            # Filter out tiny boxes\n            if box_w > min_box_size and box_h > min_box_size:\n                output.append({\n                    \"image_id\": image_id,\n                    \"x_mid\": x_center,\n                    \"y_mid\": y_center,\n                    \"w\": box_w,\n                    \"h\": box_h,\n                    \"new_class_id\": int(label) if isinstance(label, (int, float)) else label\n                })\n\n    return pd.DataFrame(output)\n\n# ✅ Display summary BEFORE WBF\nprint(\"📊 Before WBF:\")\nprint(f\"🔹 Total images: {train_df['image_id'].nunique()}\")\nprint(f\"🔹 Total labels: {len(train_df)}\\n\")\n\n# ✅ Apply WBF once\ntrain_df_wbf = apply_wbf(train_df)\n\n# ✅ Merge additional image metadata\ntrain_df_wbf = train_df_wbf.merge(\n    train_df[['image_id', 'image_path', 'width', 'height', 'source_dataset']].drop_duplicates(),\n    on='image_id',\n    how='left'\n)\n\n# ✅ Display summary AFTER WBF\nprint(\"📊 After WBF:\")\nprint(f\"✅ Total images: {train_df_wbf['image_id'].nunique()}\")\nprint(f\"✅ Total labels: {len(train_df_wbf)}\")\n\nprint(\"\\n🔍 Sample of processed DataFrame:\")\nprint(train_df_wbf.head(5))\n\nprint(\"\\n📌 Columns in WBF output:\")\nprint(train_df_wbf.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:36:50.293348Z","iopub.execute_input":"2025-05-12T07:36:50.293576Z","iopub.status.idle":"2025-05-12T07:36:59.818345Z","shell.execute_reply.started":"2025-05-12T07:36:50.293556Z","shell.execute_reply":"2025-05-12T07:36:59.817453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\n\n# ✅ Visualization function\n# ✅ Compare visualization before vs after WBF\ndef visualize_before_after(image_id, df_before, df_after, class_names=None):\n    # Get the image path from df_before\n    img_path = df_before[df_before[\"image_id\"] == image_id].iloc[0][\"image_path\"]\n    \n    # Check if the image exists\n    if not os.path.exists(img_path):\n        print(f\"❌ Image not found at {img_path}\")\n        return\n    \n    # Read and process the image\n    img = cv2.imread(img_path)\n    if img is None:\n        print(f\"❌ Failed to load image at {img_path}\")\n        return\n    \n    # Convert the image to RGB format\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    h, w = img.shape[:2]\n\n    # Set up the visualization with two subplots\n    fig, axs = plt.subplots(1, 2, figsize=(18, 8))\n    titles = [\"Before WBF\", \"After WBF\"]\n    dfs = [df_before, df_after]\n\n    # Loop through each subplot (before and after WBF)\n    for i, (ax, title, df) in enumerate(zip(axs, titles, dfs)):\n        bboxes = df[df[\"image_id\"] == image_id]\n\n        ax.imshow(img)\n        ax.set_title(f\"{title}\", fontsize=16)\n\n        # Loop through the bounding boxes and draw them on the image\n        for _, row in bboxes.iterrows():\n            x_mid = row[\"x_mid\"] * w\n            y_mid = row[\"y_mid\"] * h\n            box_w = row[\"w\"] * w\n            box_h = row[\"h\"] * h\n\n            x_min = x_mid - box_w / 2\n            y_min = y_mid - box_h / 2\n\n            # Draw rectangle for bounding box\n            rect = patches.Rectangle(\n                (x_min, y_min),\n                box_w,\n                box_h,\n                linewidth=2,\n                edgecolor='lime',\n                facecolor='none'\n            )\n            ax.add_patch(rect)\n\n            # Get class ID and label\n            class_id = int(row[\"new_class_id\"])\n            label = class_names[class_id] if class_names else str(class_id)\n            ax.text(\n                x_min, y_min - 5, label,\n                color='white',\n                fontsize=12,\n                bbox=dict(facecolor='green', alpha=0.6, edgecolor='none', pad=1)\n            )\n\n        ax.axis('off')\n\n    plt.tight_layout()\n    plt.show()\n\n# ✅ Define the class names based on your class_mapping\nclass_names = [\n    \"Aortic Enlargement\",    # ID 0\n    \"Cardiomegaly\",     # ID 1 \n    \"Lung Collapse\", #ID 2\n    \"Nodule/Mass\",   # ID 3 \n    \"Opacities/Infiltration\", # ID 4 \n    \"Pleural Conditions\"     # ID 5 \n]\n\n# ✅ Pick 5 random image_ids\nsample_ids = random.sample(list(train_df[\"image_id\"].unique()), 5)\n\n# ✅ Visualize each image before and after WBF\nfor img_id in sample_ids:\n    visualize_before_after(img_id, train_df, train_df_wbf, class_names)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:36:59.819325Z","iopub.execute_input":"2025-05-12T07:36:59.819622Z","iopub.status.idle":"2025-05-12T07:37:03.552641Z","shell.execute_reply.started":"2025-05-12T07:36:59.819595Z","shell.execute_reply":"2025-05-12T07:37:03.551818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Map class IDs to class names\nclass_id_to_name = {\n    0: \"Aortic Enlargement\",    # ID 0\n    1: \"Cardiomegaly\",          # ID 1\n    2: \"Lung Collapse\",         # ID 2\n    3: \"Nodule/Mass\",           # ID 3\n    4: \"Opacities/Infiltration\",# ID 4\n    5: \"Pleural Conditions\"     # ID 5\n}\n\n# ✅ Unique Class Distribution (Images Count) After WBF using class names\nunique_class_distribution = train_df_wbf.groupby('new_class_id')['image_id'].nunique().sort_index()\n\n# Map class IDs to class names\nunique_class_distribution = unique_class_distribution.rename(index=class_id_to_name)\n\n# Plot the distribution\nplt.figure(figsize=(12, 6))\nbars = unique_class_distribution.plot(kind='bar', color='lightgreen')\nplt.title('Unique Class Distribution (Images Count) After WBF', fontsize=16)\nplt.xlabel('Class Name', fontsize=12)\nplt.ylabel('Number of Unique Images', fontsize=12)\nplt.xticks(rotation=45)\n\n# Annotate the count on top of each bar\nfor bar in bars.patches:\n    height = bar.get_height()\n    bars.text(\n        bar.get_x() + bar.get_width() / 2, height + 50,  # Positioning the text above the bar\n        f'{height:.0f}', ha='center', va='bottom', fontsize=10\n    )\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:37:03.555251Z","iopub.execute_input":"2025-05-12T07:37:03.555495Z","iopub.status.idle":"2025-05-12T07:37:03.815249Z","shell.execute_reply.started":"2025-05-12T07:37:03.555473Z","shell.execute_reply":"2025-05-12T07:37:03.8143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Total Class Distribution (Bounding Boxes Count) After WBF using class names\ntotal_class_distribution = train_df_wbf['new_class_id'].value_counts().sort_index()\n\n# Map class IDs to class names\ntotal_class_distribution = total_class_distribution.rename(index=class_id_to_name)\n\n# Plot the distribution\nplt.figure(figsize=(12, 6))\nbars = total_class_distribution.plot(kind='bar', color='skyblue')\nplt.title('Total Class Distribution (Bounding Boxes Count) After WBF', fontsize=16)\nplt.xlabel('Class Name', fontsize=12)\nplt.ylabel('Number of Bounding Boxes', fontsize=12)\nplt.xticks(rotation=45)\n\n# Annotate the count on top of each bar\nfor bar in bars.patches:\n    height = bar.get_height()\n    bars.text(\n        bar.get_x() + bar.get_width() / 2, height + 50,  # Positioning the text above the bar\n        f'{height:.0f}', ha='center', va='bottom', fontsize=10\n    )\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:37:03.816362Z","iopub.execute_input":"2025-05-12T07:37:03.816608Z","iopub.status.idle":"2025-05-12T07:37:04.039304Z","shell.execute_reply.started":"2025-05-12T07:37:03.816574Z","shell.execute_reply":"2025-05-12T07:37:04.0384Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. ***Data Split***","metadata":{}},{"cell_type":"code","source":"# ----------------------------------------------\n# 📦 Import necessary libraries\n# ----------------------------------------------\nimport os\nimport cv2\nimport random\nimport numpy as np\nimport pandas as pd\nimport shutil\nimport matplotlib.pyplot as plt\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nfrom tqdm import tqdm\nfrom glob import glob\nfrom pathlib import Path\nfrom sklearn.model_selection import StratifiedGroupKFold\n\n# ----------------------------------------------\n# 📦 Split original data BEFORE augmentation\n# ----------------------------------------------\n\n# Assuming your original dataframe is `train_df_wbf`\nbalanced_df_multi = train_df_wbf.groupby('image_id')['new_class_id'].agg(lambda x: list(set(x))).reset_index()\ntrain_df_wbf = train_df_wbf.merge(balanced_df_multi, on='image_id', suffixes=(\"\", \"_multi\"))\ntrain_df_wbf = train_df_wbf.loc[:, ~train_df_wbf.columns.duplicated()]\ntrain_df_wbf['multi_class_str'] = train_df_wbf['new_class_id_multi'].apply(lambda x: str(sorted(x)))\n\nsgkf = StratifiedGroupKFold(n_splits=4, shuffle=True, random_state=42)\n\nfor train_idx, val_idx in sgkf.split(train_df_wbf, train_df_wbf['multi_class_str'], groups=train_df_wbf['image_id']):\n    train_df_split = train_df_wbf.iloc[train_idx].reset_index(drop=True)\n    val_df_split = train_df_wbf.iloc[val_idx].reset_index(drop=True)\n    break\n\n# Drop extra columns not needed after split\ntrain_df_split.drop(columns=['new_class_id_multi', 'multi_class_str'], inplace=True)\nval_df_split.drop(columns=['new_class_id_multi', 'multi_class_str'], inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:37:04.040236Z","iopub.execute_input":"2025-05-12T07:37:04.040483Z","iopub.status.idle":"2025-05-12T07:37:06.06584Z","shell.execute_reply.started":"2025-05-12T07:37:04.040463Z","shell.execute_reply":"2025-05-12T07:37:06.065072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def check_bboxes_validity(df):\n    valid_bbox_count = 0\n    invalid_bbox_count = 0\n\n    valid_yolo_count = 0\n    invalid_yolo_count = 0\n\n    total = len(df)\n\n    for _, row in df.iterrows():\n        # YOLO-format components\n        x_center, y_center = row['x_mid'], row['y_mid']\n        box_width, box_height = row['w'], row['h']\n        \n        # Convert YOLO to corner format\n        x_min = x_center - box_width / 2\n        y_min = y_center - box_height / 2\n        x_max = x_center + box_width / 2\n        y_max = y_center + box_height / 2\n\n        # Check corner format validity and normalized range\n        if 0 <= x_min < x_max <= 1 and 0 <= y_min < y_max <= 1:\n            valid_bbox_count += 1\n        else:\n            invalid_bbox_count += 1\n\n        # Check YOLO format validity (size > 0)\n        if box_width > 0 and box_height > 0:\n            valid_yolo_count += 1\n        else:\n            invalid_yolo_count += 1\n\n    # Print summary\n    print(f\"\\n🔍 Total bounding boxes: {total}\")\n    print(f\"✅ Valid (corner format): {valid_bbox_count} ({valid_bbox_count / total:.2%})\")\n    print(f\"❌ Invalid (corner format): {invalid_bbox_count} ({invalid_bbox_count / total:.2%})\")\n    print(f\"✅ Valid (YOLO format): {valid_yolo_count} ({valid_yolo_count / total:.2%})\")\n    print(f\"❌ Invalid (YOLO format): {invalid_yolo_count} ({invalid_yolo_count / total:.2%})\")\n\n    # Optional warning\n    if invalid_bbox_count > 0 or invalid_yolo_count > 0:\n        print(\"⚠️ Warning: Some bounding boxes are invalid. Please review the data.\")\n\n# ✅ Use it\nprint(\"Checking bounding box validity...\")\ncheck_bboxes_validity(train_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:37:06.066858Z","iopub.execute_input":"2025-05-12T07:37:06.067223Z","iopub.status.idle":"2025-05-12T07:37:07.960528Z","shell.execute_reply.started":"2025-05-12T07:37:06.067193Z","shell.execute_reply":"2025-05-12T07:37:07.959587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def count_bboxes_formats(df):\n    # Convert YOLO to corner format\n    df = df.copy()\n    df['x_min'] = df['x_mid'] - df['w'] / 2\n    df['y_min'] = df['y_mid'] - df['h'] / 2\n    df['x_max'] = df['x_mid'] + df['w'] / 2\n    df['y_max'] = df['y_mid'] + df['h'] / 2\n\n    # Check YOLO validity: all components must be finite and w, h > 0\n    yolo_invalid = (\n        df[['x_mid', 'y_mid', 'w', 'h']].isna().any(axis=1) |\n        (df['w'] <= 0) | (df['h'] <= 0)\n    )\n\n    # Check Corner validity: x_max > x_min, y_max > y_min\n    corner_invalid = (\n        df[['x_min', 'y_min', 'x_max', 'y_max']].isna().any(axis=1) |\n        (df['x_max'] <= df['x_min']) | (df['y_max'] <= df['y_min'])\n    )\n\n    # Count\n    total = len(df)\n    valid_yolo = (~yolo_invalid).sum()\n    valid_corner = (~corner_invalid).sum()\n\n    print(f\"🔎 Total bounding boxes: {total}\")\n    print(f\"✅ Valid YOLO format: {valid_yolo} ({valid_yolo / total:.2%})\")\n    print(f\"❌ Invalid YOLO format: {total - valid_yolo} ({(total - valid_yolo) / total:.2%})\")\n    print(f\"✅ Valid Corner format: {valid_corner} ({valid_corner / total:.2%})\")\n    print(f\"❌ Invalid Corner format: {total - valid_corner} ({(total - valid_corner) / total:.2%})\")\n\n# ✅ Run the check\ncount_bboxes_formats(train_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:37:07.961462Z","iopub.execute_input":"2025-05-12T07:37:07.961674Z","iopub.status.idle":"2025-05-12T07:37:07.980687Z","shell.execute_reply.started":"2025-05-12T07:37:07.961656Z","shell.execute_reply":"2025-05-12T07:37:07.97992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport shutil\nimport pandas as pd\nfrom tqdm import tqdm\nfrom pathlib import Path\nfrom glob import glob\nimport albumentations as A\n\n# ----------------------------------------------\n# 📦 Fix paths for 'chestxrayabnormalities' images\n# ----------------------------------------------\ndef fix_chestxrayabnormalities_paths(df, base_dir):\n    df = df.copy()\n    mask = df['source_dataset'] == 'chestxrayabnormalities'\n\n    image_files = glob(os.path.join(base_dir, '*.jpg'))\n    stem_to_path = {Path(f).stem.split('_jpg')[0]: f for f in image_files}\n\n    def correct_path(row):\n        if row['source_dataset'] != 'chestxrayabnormalities':\n            return row['image_path']\n        image_id = Path(row['image_path']).stem\n        return stem_to_path.get(image_id, row['image_path'])  # fallback\n\n    df.loc[mask, 'image_path'] = df[mask].apply(correct_path, axis=1)\n    return df\n\n# ----------------------------------------------\n# 📦 Function to define augmentations\n# ----------------------------------------------\ndef get_chest_xray_augmentations():\n    return A.Compose([\n        A.RandomBrightnessContrast(brightness_limit=0.1, contrast_limit=0.1, p=0.5),\n        A.RandomGamma(gamma_limit=(95, 105), p=0.5),\n        A.Rotate(limit=5, p=0.5),\n        A.ShiftScaleRotate(shift_limit=0.01, scale_limit=0.05, rotate_limit=3, p=0.5, border_mode=0),\n        A.CLAHE(clip_limit=2.0, tile_grid_size=(8, 8), p=0.3),\n        A.GaussNoise(var_limit=(5.0, 10.0), p=0.2),\n    ], bbox_params=A.BboxParams(format='pascal_voc', label_fields=['labels']))\n\n# ----------------------------------------------\n# 📦 Apply augmentation\n# ----------------------------------------------\ndef apply_augmentation(image, bboxes, labels):\n    transform = get_chest_xray_augmentations()\n    augmented = transform(image=image, bboxes=bboxes, labels=labels)\n    return augmented['image'], augmented['bboxes']\n\n# ----------------------------------------------\n# 📦 Identify low-frequency classes\n# ----------------------------------------------\ndef identify_low_freq_classes(df, moderate_threshold=0.165, strong_threshold=0.1):\n    class_counts = df['new_class_id'].value_counts(normalize=True)\n    strong_freq_classes = class_counts[class_counts < strong_threshold].index.tolist()\n    moderate_freq_classes = class_counts[(class_counts >= strong_threshold) & (class_counts < moderate_threshold)].index.tolist()\n\n    print(f\"Strong frequency classes (<{strong_threshold}): {strong_freq_classes}\")\n    print(f\"Moderate frequency classes (<{moderate_threshold}): {moderate_freq_classes}\")\n    return moderate_freq_classes, strong_freq_classes\n\n# ----------------------------------------------\n# 📦 Save one image\n# ----------------------------------------------\ndef save_image(save_path, image):\n    cv2.imwrite(save_path, image)\n\n# ----------------------------------------------\n# 📦 Augment and balance training data\n# ----------------------------------------------\ndef augment_and_balance_data(train_df, save_augmented_dir='augmented_images', target_image_count=3000):\n    train_df = train_df[train_df['source_dataset'] != 'chestxrayabnormalities'].copy()\n    augmented_data = []\n    augmented_image_ids = set()\n\n    os.makedirs(save_augmented_dir, exist_ok=True)\n\n    class_counts = train_df['new_class_id'].value_counts()\n    print(f\"[INFO] Class distribution before augmentation:\")\n    print(class_counts)\n\n    temp_df = train_df.copy()\n    print(f\"[INFO] Targeting {target_image_count} images per class.\")\n\n    for class_id in tqdm(class_counts.index, desc=\"Augmenting Classes\"):\n        current_count = (temp_df['new_class_id'] == class_id).sum()\n        if current_count >= target_image_count:\n            continue\n\n        augment_needed = target_image_count - current_count\n        class_group = train_df[train_df['new_class_id'] == class_id]\n        sampled_group = class_group.sample(n=augment_needed, replace=True, random_state=42)\n\n        for idx, (image_id, group) in enumerate(sampled_group.groupby('image_id')):\n            image_path = group.iloc[0]['image_path']\n            image = cv2.imread(image_path)\n            if image is None:\n                print(f\"[WARN] Failed to load image {image_path}. Skipping.\")\n                continue\n\n            h_img, w_img = image.shape[:2]\n            bboxes, labels = [], []\n\n            for _, row in group.iterrows():\n                box_width, box_height = row['w'] * w_img, row['h'] * h_img\n                x_center, y_center = row['x_mid'] * w_img, row['y_mid'] * h_img\n                x_min = x_center - box_width / 2\n                y_min = y_center - box_height / 2\n                x_max = x_center + box_width / 2\n                y_max = y_center + box_height / 2\n\n                if x_min >= x_max or y_min >= y_max or box_width <= 0 or box_height <= 0:\n                    continue\n\n                bboxes.append((x_min, y_min, x_max, y_max))\n                labels.append(row['new_class_id'])\n\n            if not bboxes:\n                continue\n\n            try:\n                augmented_image, augmented_bboxes = apply_augmentation(image.copy(), bboxes, labels)\n            except Exception as e:\n                print(f\"[ERROR] Augmentation failed for {image_id}: {e}\")\n                continue\n\n            h_aug, w_aug = augmented_image.shape[:2]\n            augmented_image_id = f\"{image_id}_aug_{len(augmented_image_ids)}\"\n\n            if augmented_image_id not in augmented_image_ids:\n                augmented_image_ids.add(augmented_image_id)\n                save_path = os.path.join(save_augmented_dir, f\"{augmented_image_id}.jpg\")\n                save_image(save_path, augmented_image)\n\n                for (x_min, y_min, x_max, y_max), class_id_aug in zip(augmented_bboxes, labels):\n                    x_min = max(0, min(x_min, w_aug))\n                    x_max = max(0, min(x_max, w_aug))\n                    y_min = max(0, min(y_min, h_aug))\n                    y_max = max(0, min(y_max, h_aug))\n\n                    if x_max <= x_min or y_max <= y_min:\n                        continue\n\n                    x_center = ((x_min + x_max) / 2) / w_aug\n                    y_center = ((y_min + y_max) / 2) / h_aug\n                    width = (x_max - x_min) / w_aug\n                    height = (y_max - y_min) / h_aug\n\n                    augmented_data.append({\n                        'image_id': augmented_image_id,\n                        'image_path': save_path,\n                        'new_class_id': class_id_aug,\n                        'x_mid': x_center,\n                        'y_mid': y_center,\n                        'w': width,\n                        'h': height\n                    })\n\n    print(\"[✅] Augmented images saved!\")\n\n    augmented_df = pd.DataFrame(augmented_data)\n    balanced_df = pd.concat([train_df, augmented_df], ignore_index=True)\n    balanced_df = balanced_df.sample(frac=1, random_state=42).reset_index(drop=True)\n\n    print(f\"[✅] Balancing complete. Class distribution:\")\n    print(balanced_df['new_class_id'].value_counts())\n\n    return balanced_df\n\n# ----------------------------------------------\n# 📦 Prepare YOLO labels\n# ----------------------------------------------\ndef prepare_yolo_labels(df, image_dest_dir, label_dest_dir):\n    df = df.drop_duplicates(subset=['image_id', 'new_class_id', 'x_mid', 'y_mid', 'w', 'h'])\n    os.makedirs(image_dest_dir, exist_ok=True)\n    os.makedirs(label_dest_dir, exist_ok=True)\n\n    for image_id, group in tqdm(df.groupby('image_id'), desc=f\"Processing {image_dest_dir}\"):\n        image_path = group.iloc[0]['image_path']\n        image_name = os.path.basename(image_path)\n        label_file = Path(label_dest_dir) / f\"{Path(image_name).stem}.txt\"\n        image_target = Path(image_dest_dir) / image_name\n\n        if not image_target.exists():\n            shutil.copy(image_path, image_target)\n\n        group_sorted = group.sort_values(by='new_class_id')\n\n        with open(label_file, 'w') as f:\n            for _, row in group_sorted.iterrows():\n                f.write(f\"{row['new_class_id']} {row['x_mid']:.6f} {row['y_mid']:.6f} {row['w']:.6f} {row['h']:.6f}\\n\")\n\n# ----------------------------------------------\n# 📦 Deduplicate YOLO labels\n# ----------------------------------------------\ndef deduplicate_yolo_labels(label_dir):\n    label_paths = glob(os.path.join(label_dir, \"*.txt\"))\n    total_files = len(label_paths)\n    deduplicated_count = 0\n\n    for path in label_paths:\n        with open(path, 'r') as f:\n            lines = f.readlines()\n        original_count = len(lines)\n        deduped = list(set([line.strip() for line in lines]))\n\n        if len(deduped) < original_count:\n            deduplicated_count += 1\n            with open(path, 'w') as f:\n                f.write('\\n'.join(deduped) + '\\n')\n\n    print(f\"✅ Deduplicated {deduplicated_count}/{total_files} files in: {label_dir}\")\n\n# ----------------------------------------------\n# 📦 Full Processing Flow\n# ----------------------------------------------\n# Assuming train_df_split and val_df_split are already defined\n\n# 1. 🩺 Fix validation image paths (for chestxrayabnormalities)\nval_df_split = fix_chestxrayabnormalities_paths(val_df_split, '/kaggle/input/chestxrayabnormalities/train/images')\n\n# 2. 📈 Augment training set only\nbalanced_train_df = augment_and_balance_data(train_df_split)\n\n# 3. 📄 Prepare YOLO labels\nprepare_yolo_labels(balanced_train_df, 'data/images/train', 'data/labels/train')\nprepare_yolo_labels(val_df_split, 'data/images/val', 'data/labels/val')\n\n# 4. 🧹 Deduplicate labels\ndeduplicate_yolo_labels('data/labels/train')\ndeduplicate_yolo_labels('data/labels/val')\n\n# 5. 🧼 Clean up temporary files\nshutil.rmtree('augmented_images')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:37:07.981547Z","iopub.execute_input":"2025-05-12T07:37:07.981779Z","iopub.status.idle":"2025-05-12T07:41:53.945684Z","shell.execute_reply.started":"2025-05-12T07:37:07.981758Z","shell.execute_reply":"2025-05-12T07:41:53.944168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_id_to_name = {\n    0: \"Aortic Enlargement\",\n    1: \"Cardiomegaly\",\n    2: \"Lung Collapse\",\n    3: \"Nodule/Mass\",\n    4: \"Opacities/Infiltration\",\n    5: \"Pleural Conditions\"\n}\n\n# Bounding box count per class\nbbox_counts = balanced_train_df['new_class_id'].value_counts().sort_index()\n\n# Image count per class\nimage_counts = balanced_train_df.groupby('new_class_id')['image_id'].nunique().sort_index()\n\n# Combine both into a DataFrame\nsummary_df = pd.DataFrame({\n    'class_id': bbox_counts.index,\n    'class_name': [class_id_to_name[i] for i in bbox_counts.index],\n    'bbox_count': bbox_counts.values,\n    'unique_image_count': image_counts.values\n})\n\nprint(summary_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:41:53.946785Z","iopub.execute_input":"2025-05-12T07:41:53.947035Z","iopub.status.idle":"2025-05-12T07:41:53.961376Z","shell.execute_reply.started":"2025-05-12T07:41:53.947015Z","shell.execute_reply":"2025-05-12T07:41:53.960449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport random\nfrom PIL import Image\nfrom pathlib import Path\n\n# Define the path to your augmented image directory\naugmented_image_dir = Path('data/images/train')\n\n# Get list of all .png image files\naugmented_image_files = sorted([f for f in augmented_image_dir.glob('*') if f.suffix.lower() in ['.png', '.jpg', '.jpeg']])\n\n# Limit to available number of files\nnum_samples = min(5, len(augmented_image_files))\nif num_samples == 0:\n    print(\"No images found.\")\nelse:\n    # Optional: for reproducibility\n    random.seed(42)\n    sample_images = random.sample(augmented_image_files, num_samples)\n\n    # Plot sampled images\n    fig, axes = plt.subplots(1, num_samples, figsize=(3 * num_samples, 6))\n\n    if num_samples == 1:\n        axes = [axes]\n\n    for ax, img_path in zip(axes, sample_images):\n        img = Image.open(img_path).convert(\"RGB\")\n        ax.imshow(img)\n        ax.axis('off')\n        ax.set_title(img_path.name)\n\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:41:53.962296Z","iopub.execute_input":"2025-05-12T07:41:53.962655Z","iopub.status.idle":"2025-05-12T07:41:54.984342Z","shell.execute_reply.started":"2025-05-12T07:41:53.962619Z","shell.execute_reply":"2025-05-12T07:41:54.983495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport os\n\ndef check_channels(image_dir):\n    for f in os.listdir(image_dir):\n        if f.lower().endswith(('.jpg', '.png', '.jpeg')):\n            img = cv2.imread(os.path.join(image_dir, f))\n            if img is not None and img.shape[2] != 3:\n                print(f\"{f} is not RGB\")\n\ncheck_channels('/kaggle/working/data/images/train')\ncheck_channels('/kaggle/working/data/images/val')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:41:54.985256Z","iopub.execute_input":"2025-05-12T07:41:54.985531Z","iopub.status.idle":"2025-05-12T07:43:04.393099Z","shell.execute_reply.started":"2025-05-12T07:41:54.985505Z","shell.execute_reply":"2025-05-12T07:43:04.392332Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. ***data.yaml file & YOLO12 Training***","metadata":{}},{"cell_type":"code","source":"import cv2\nimport os\n\ndef convert_grayscale_to_rgb_in_place(image_dir):\n    \"\"\"\n    Convert grayscale or single-channel images to 3-channel RGB for YOLO compatibility.\n    Overwrites the images in place.\n    \"\"\"\n    image_files = [f for f in os.listdir(image_dir) if f.lower().endswith(('.png', '.jpg', '.jpeg'))]\n    total = len(image_files)\n    converted = 0\n\n    print(f\"Processing {total} images in: {image_dir}\")\n\n    for idx, image_file in enumerate(image_files, 1):\n        image_path = os.path.join(image_dir, image_file)\n        img = cv2.imread(image_path)\n\n        if img is None:\n            print(f\"⚠️ Failed to read: {image_file}\")\n            continue\n\n        # Skip if image is already RGB (3 channels)\n        if len(img.shape) == 3 and img.shape[2] == 3:\n            continue\n\n        # Convert grayscale to RGB\n        img_rgb = cv2.cvtColor(img, cv2.COLOR_GRAY2RGB)\n        cv2.imwrite(image_path, img_rgb)\n        converted += 1\n\n        if idx % 100 == 0 or idx == total:\n            print(f\"[{idx}/{total}] processed, {converted} converted\", flush=True)\n\n    print(f\"✅ Done. {converted} images converted in: {image_dir}\\n\")\n\n\n# Directories\ntrain_dir = '/kaggle/working/data/images/train'\nval_dir = '/kaggle/working/data/images/val'\n\n# Convert grayscale images to 3-channel RGB\nconvert_grayscale_to_rgb_in_place(train_dir)\nconvert_grayscale_to_rgb_in_place(val_dir)\n\n# Class names for YOLO\nclass_names = [\n    'Aortic Enlargement',\n    'Cardiomegaly',\n    'Lung Collapse',\n    'Nodule/Mass',\n    'Opacities/Infiltration',\n    'Pleural Conditions'\n]\n\n# Correctly format the names\nnames_yaml = '\\n'.join([f\"  - {name}\" for name in class_names])\n\n# Create YOLO data.yaml dynamically\ndata_yaml = f\"\"\"train: {train_dir}\nval: {val_dir}\n\nnc: {len(class_names)}\nnames:\n{names_yaml}\n\"\"\"\n\n# Save data.yaml\nyaml_path = 'data.yaml'\ntry:\n    with open(yaml_path, 'w') as f:\n        f.write(data_yaml)\n    print(f\"✅ data.yaml file created at {yaml_path}\")\nexcept Exception as e:\n    print(f\"❌ Failed to create data.yaml: {e}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:43:04.393867Z","iopub.execute_input":"2025-05-12T07:43:04.394177Z","iopub.status.idle":"2025-05-12T07:44:13.989342Z","shell.execute_reply.started":"2025-05-12T07:43:04.394149Z","shell.execute_reply":"2025-05-12T07:44:13.988417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Check available GPUs\nnum_gpus = torch.cuda.device_count()\nprint(f\"Available GPUs: {num_gpus}\")\nfor i in range(num_gpus):\n    print(f\"GPU {i}: {torch.cuda.get_device_name(i)}\")\n\n# ✅ Set device for training\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(f\"Using {device} for training\")\n\n# ✅ Optimize CUDA memory allocation\ntorch.cuda.empty_cache()\nos.environ[\"PYTORCH_CUDA_ALLOC_CONF\"] = \"expandable_segments:True\"\n\n# ✅ Load YOLOv12-M model\nmodel = YOLO(\"yolo12m.pt\").to(device)\n\n# ✅ Train the model\ntrain_results = model.train(\n    data=\"/kaggle/working/data.yaml\",\n    epochs=110,\n    batch=22,\n    imgsz=640,\n    device=\"auto\",     # This will still use GPU automatically\n    half=True,\n    amp=True,\n    workers=5,\n    cache=False,\n    save=True,\n    save_period=10,\n    patience=15,\n    project=\"yolov12-training\",\n    name=\"yolo12m-vinbigdata\",\n    exist_ok=True,\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T07:44:13.990227Z","iopub.execute_input":"2025-05-12T07:44:13.990529Z","iopub.status.idle":"2025-05-12T18:44:55.312329Z","shell.execute_reply.started":"2025-05-12T07:44:13.990502Z","shell.execute_reply":"2025-05-12T18:44:55.310534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Check training results\nprint(\"✅ Training results:\")\nprint(train_results)\n\n# ✅ Validate the model\nmetrics = model.val()\n\n# ✅ Print evaluation metrics\nprint(\"Evaluation metrics:\")\nprint(metrics)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T18:51:51.793383Z","iopub.execute_input":"2025-05-12T18:51:51.79376Z","iopub.status.idle":"2025-05-12T18:52:32.205254Z","shell.execute_reply.started":"2025-05-12T18:51:51.793717Z","shell.execute_reply":"2025-05-12T18:52:32.204126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!zip -r yolov12-training.zip /kaggle/working/yolov12-training/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T18:57:00.003421Z","iopub.execute_input":"2025-05-12T18:57:00.003791Z","iopub.status.idle":"2025-05-12T18:57:49.599887Z","shell.execute_reply.started":"2025-05-12T18:57:00.003741Z","shell.execute_reply":"2025-05-12T18:57:49.59903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink\n# Provide a clickable download link\nFileLink(r'yolov12-training.zip')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T18:59:37.283605Z","iopub.execute_input":"2025-05-12T18:59:37.283924Z","iopub.status.idle":"2025-05-12T18:59:37.290239Z","shell.execute_reply.started":"2025-05-12T18:59:37.2839Z","shell.execute_reply":"2025-05-12T18:59:37.289337Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ***6. Mode & 90th percentile***","metadata":{}},{"cell_type":"code","source":"!pip install ultralytics\nfrom ultralytics import YOLO","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_path = \"/kaggle/input/mymodel/pytorch/yolo12/1/best.pt\"\nmodel = YOLO(model_path)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\nimport glob\nfrom collections import defaultdict, Counter\n\n# Move model to GPU\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.to(device)\n\n# Get all image file paths\nimage_dir = \"/kaggle/input/vinbigdata-1024-image-dataset/vinbigdata/test/\"\nimage_paths = sorted(glob.glob(image_dir + \"*.png\"))  # Adjust extension if needed\n\nbatch_size = 16  # Adjust based on GPU memory\nall_results = []\n\n# Run inference in batches\nfor i in range(0, len(image_paths), batch_size):\n    batch = image_paths[i : i + batch_size]  # Get batch of image paths\n    batch_results = model(batch, conf=0.1)  # Run inference\n    all_results.extend(batch_results)\n\n# Collect confidence scores by class\nconf_scores = defaultdict(list)\n\nfor result in all_results:\n    if result.boxes is not None:  # Ensure detections exist\n        for det in result.boxes.to(device):  # Keep tensors on GPU\n            cls = int(det.cls.item())  # Get class ID\n            conf = float(det.conf.item())  # Get confidence score\n            class_name = model.names[cls] if hasattr(model, \"names\") else str(cls)\n            conf_scores[class_name].append(conf)\n\n# Calculate the mode (most common confidence score) and 90th percentile for each class\nclass_conf_stats = {}\n\nfor cls, scores in conf_scores.items():\n    # Calculate the mode\n    mode_conf = Counter(scores).most_common(1)[0][0]\n    \n    # Calculate the 90th percentile\n    percentile_90 = np.percentile(scores, 90)\n    \n    class_conf_stats[cls] = {\n        \"mode\": mode_conf,\n        \"90th_percentile\": percentile_90\n    }\n\nprint(class_conf_stats)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# modes & 90th percentile for each class\nclass_conf_data = {\n    'Cardiac & Vascular': {'mode': 0.131, '90th_percentile': 0.705},\n    'Pleural Abnormalities': {'mode': 0.115, '90th_percentile': 0.389},\n    'Lung Collapse': {'mode': 0.111, '90th_percentile': 0.507},\n    'Fibrosis & ILD': {'mode': 0.337, '90th_percentile': 0.635},\n    'Nodule/Mass or Other Lesion': {'mode': 0.208, '90th_percentile': 0.524},\n    'Lung Opacity': {'mode': 0.195, '90th_percentile': 0.515}\n}\n\nadjusted_thresholds = {}\n\nfor cls, values in class_conf_data.items():\n    mode = values['mode']\n    perc90 = values['90th_percentile']\n\n    if mode < 0.2:\n        # Use the 75th percentile if mode is too low\n        new_threshold = np.percentile([mode, perc90], 75)\n    elif mode >= 0.3:\n        # Use the mode directly if reasonable\n        new_threshold = mode\n    else:\n        # Use a weighted average if 90th percentile is much higher\n        new_threshold = (0.7 * mode) + (0.3 * perc90)\n\n    adjusted_thresholds[cls] = round(new_threshold, 3)\n\nprint(adjusted_thresholds)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nimport torch\nfrom PIL import Image\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\nfrom ultralytics import YOLO  \n\n# ✅ Force inline display in Kaggle\n%matplotlib inline  \n\n# Load YOLO model\nmodel_path = \"/kaggle/input/mymodel/pytorch/yolo12/1/best.pt\"\nmodel = YOLO(model_path)\n\n# Confidence thresholds\nadjusted_thresholds = {\n    'Cardiac & Vascular': 0.562,\n    'Pleural Abnormalities': 0.32,\n    'Lung Collapse': 0.408,\n    'Fibrosis & ILD': 0.337,\n    'Nodule/Mass or Other Lesion': 0.303,\n    'Lung Opacity': 0.435\n}\n\n# Unique colors per class\nclass_colors = {\n    'Cardiac & Vascular': (255, 0, 0),\n    'Pleural Abnormalities': (0, 255, 0),\n    'Lung Collapse': (0, 0, 255),\n    'Fibrosis & ILD': (255, 255, 0),\n    'Nodule/Mass or Other Lesion': (255, 165, 0),\n    'Lung Opacity': (128, 0, 128)\n}\n\n# Dataset path\ndataset_path = Path(\"/kaggle/input/testing/\")\n\n# Function to preprocess image\ndef preprocess_image(image_path):\n    img = Image.open(image_path).convert(\"RGB\")\n    return img\n\n# Process images\nfor image_path in dataset_path.glob(\"*.*\"):\n    if image_path.suffix.lower() in [\".png\", \".jpg\", \".jpeg\"]:\n        processed_img = preprocess_image(image_path)\n        img_array = np.array(processed_img)  \n        img_height, img_width = img_array.shape[:2]  \n\n        # Run inference\n        results = model(processed_img)[0]  \n\n        # Extract boxes, confidences, and classes\n        boxes = results.boxes.xyxy.cpu().numpy()\n        confidences = results.boxes.conf.cpu().numpy()\n        classes = results.boxes.cls.cpu().numpy().astype(int)\n\n        filtered_boxes, filtered_classes, filtered_confidences = [], [], []\n\n        for i in range(len(classes)):\n            class_idx = classes[i]\n            class_name = model.names[class_idx]\n            conf = confidences[i]\n\n            class_threshold = adjusted_thresholds.get(class_name, 0.1)\n            if conf >= class_threshold:\n                filtered_boxes.append(boxes[i])\n                filtered_classes.append(class_name)\n                filtered_confidences.append(conf)\n\n        # If valid detections exist\n        if filtered_boxes:\n            image = np.array(processed_img)  \n\n            for i in range(len(filtered_boxes)):\n                x1, y1, x2, y2 = map(int, filtered_boxes[i])\n\n                # Assign unique color\n                color = class_colors.get(filtered_classes[i], (255, 255, 255))  \n\n                # Dynamic font scaling\n                font_scale = max(0.5, min(img_width, img_height) / 600)  # Adjusted for image size\n                thickness = max(2, int(font_scale * 2))  # Bold text by increasing thickness\n\n                # Draw bounding box\n                cv2.rectangle(image, (x1, y1), (x2, y2), color, thickness)  \n\n                # Text label (No background)\n                label = f\"{filtered_classes[i]} ({filtered_confidences[i]:.2f})\"\n                \n                # Draw text\n                cv2.putText(image, label, (x1, y1 - 5), cv2.FONT_HERSHEY_SIMPLEX, font_scale, color, thickness)\n\n            # ✅ Convert BGR to RGB for proper display\n            image_rgb = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n            # ✅ Ensure images appear properly\n            plt.figure(figsize=(8, 8))\n            plt.imshow(image_rgb)  # Show the image in RGB format\n            plt.axis(\"off\")\n            plt.title(f\"Detections for {image_path.name}\")\n            plt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%matplotlib inline\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}