{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":11215393,"datasetId":6799157,"databundleVersionId":11623300},{"sourceType":"datasetVersion","sourceId":2057341,"datasetId":1232864,"databundleVersionId":2097467},{"sourceType":"datasetVersion","sourceId":8785422,"datasetId":5281464,"databundleVersionId":8941916},{"sourceType":"datasetVersion","sourceId":1799839,"datasetId":1069682,"databundleVersionId":1837296},{"sourceType":"datasetVersion","sourceId":1799615,"datasetId":1069544,"databundleVersionId":1837072},{"sourceType":"modelInstanceVersion","sourceId":309588,"databundleVersionId":11623012,"modelInstanceId":262719}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. ***Libraries Installation & Importing*** ","metadata":{}},{"cell_type":"code","source":"!pip install ultralytics\n!pip install torchxrayvision\n!pip install pydicom Pillow\n!pip install scikit-image\n!pip install tqdm --upgrade\n!pip install scikit-learn\n!pip install -q ensemble-boxes","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:08:39.441309Z","iopub.execute_input":"2025-04-08T12:08:39.441605Z","iopub.status.idle":"2025-04-08T12:09:06.503834Z","shell.execute_reply.started":"2025-04-08T12:08:39.441575Z","shell.execute_reply":"2025-04-08T12:09:06.502924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ─────────────────────────────\n# ✅ Standard libraries\n# ─────────────────────────────\nimport os\nimport gc\nimport ast\nimport zipfile\nimport shutil\nimport random\nimport pprint\nimport warnings\nfrom glob import glob\nfrom collections import Counter\nfrom concurrent.futures import ThreadPoolExecutor\n\n# ─────────────────────────────\n# ✅ Data handling\n# ─────────────────────────────\nimport numpy as np\nimport pandas as pd\nimport yaml\nfrom tqdm.autonotebook import tqdm\n\n# ─────────────────────────────\n# ✅ Image handling & visualization\n# ─────────────────────────────\nimport cv2\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport seaborn as sns\nfrom IPython.display import display, FileLink\nimport pydicom\n\n# ─────────────────────────────\n# ✅ Scientific image processing\n# ─────────────────────────────\nimport skimage.io\nimport skimage.transform\nimport albumentations as A\n\n# ─────────────────────────────\n# ✅ Machine learning & utilities\n# ─────────────────────────────\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom sklearn.manifold import TSNE\n\n# ─────────────────────────────\n# ✅ Deep learning\n# ─────────────────────────────\nimport torch\nimport torch.nn.functional as F\nimport torchvision\nimport torchvision.transforms as T\nfrom torch.utils.data import DataLoader, Dataset\nimport torchxrayvision as xrv  # For X-ray-specific processing\n\n# ─────────────────────────────\n# ✅ Object Detection (YOLO & WBF)\n# ─────────────────────────────\nfrom ultralytics import YOLO\nfrom ensemble_boxes import weighted_boxes_fusion","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:06.506043Z","iopub.execute_input":"2025-04-08T12:09:06.506303Z","iopub.status.idle":"2025-04-08T12:09:17.069Z","shell.execute_reply.started":"2025-04-08T12:09:06.506283Z","shell.execute_reply":"2025-04-08T12:09:17.068034Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Data Preparation, Bounding Box Visualization, and Class Distribution","metadata":{}},{"cell_type":"code","source":"# ✅ Load dataset\nlabel_data_file = \"/kaggle/input/vinbigdata-1024-image-dataset/vinbigdata/train.csv\"\ntrain_df = pd.read_csv(label_data_file)\n\n# ✅ Add image_path column\ntrain_df['image_path'] = '/kaggle/input/vinbigdata-1024-image-dataset/vinbigdata/train/' + train_df.image_id + '.png'\n\n# ✅ Remove class 14 (No Finding) and class 2 (Calcification) completely\ntrain_df = train_df[~train_df.class_id.isin([14, 2])].reset_index(drop=True)\n\n# ✅ Print remaining images to confirm\nprint(f\"✅ Number of images remaining: {train_df['image_id'].nunique()}\")\n\n# ✅ Convert VinBigData bbox format to YOLO format\ntrain_df['x_mid'] = (train_df['x_min'] + train_df['x_max']) / (2 * train_df['width'])\ntrain_df['y_mid'] = (train_df['y_min'] + train_df['y_max']) / (2 * train_df['height'])\ntrain_df['w'] = (train_df['x_max'] - train_df['x_min']) / train_df['width']\ntrain_df['h'] = (train_df['y_max'] - train_df['y_min']) / train_df['height']\n\ntrain_df['source_dataset'] = 'vinbig'\n\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:17.070596Z","iopub.execute_input":"2025-04-08T12:09:17.071155Z","iopub.status.idle":"2025-04-08T12:09:17.351051Z","shell.execute_reply.started":"2025-04-08T12:09:17.071124Z","shell.execute_reply":"2025-04-08T12:09:17.350285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Define the class_name_to_id mapping\nclass_name_to_id = {\n    \"Aortic enlargement\": 0,\n    \"Cardiomegaly\": 2,  \n    \"Consolidation\": 3,\n    \"ILD\": 4,\n    \"Infiltration\": 5,\n    \"Lung Opacity\": 6,\n    \"Nodule/Mass\": 7,\n    \"Other lesion\": 8,\n    \"Pleural effusion\": 9,\n    \"Pleural thickening\": 10,\n    \"Pneumothorax\": 11,\n    \"Pulmonary fibrosis\": 12,\n    \"Atelectasis\": 1\n}\n\n# ✅ Print final dataset details\nprint(f\"✅ Number of unique images after final merge: {train_df['image_id'].nunique()}\")\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:17.351828Z","iopub.execute_input":"2025-04-08T12:09:17.352042Z","iopub.status.idle":"2025-04-08T12:09:17.373488Z","shell.execute_reply.started":"2025-04-08T12:09:17.352025Z","shell.execute_reply":"2025-04-08T12:09:17.372505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_bboxes(image_paths, bboxes_list, labels_list, image_sizes, num_images=5):\n    num_images = min(num_images, len(image_paths))  \n    fig, axes = plt.subplots(1, num_images, figsize=(50, 30))\n\n    if num_images == 1:\n        axes = [axes]\n\n    for idx in range(num_images):\n        image_path, bboxes, labels, (orig_width, orig_height) = (\n            image_paths[idx], bboxes_list[idx], labels_list[idx], image_sizes[idx])\n\n        image = cv2.imread(image_path)\n        if image is None:\n            print(f\"⚠️ Error: Could not read {image_path}\")\n            continue\n\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)  \n        height, width, _ = image.shape  \n\n        # ✅ Rescale bounding boxes\n        for i in range(len(bboxes)):\n            x_min, y_min, x_max, y_max = bboxes[i]\n            bboxes[i] = [\n                int((x_min / orig_width) * width),\n                int((y_min / orig_height) * height),\n                int((x_max / orig_width) * width),\n                int((y_max / orig_height) * height),\n            ]\n            cv2.rectangle(image, (bboxes[i][0], bboxes[i][1]), (bboxes[i][2], bboxes[i][3]), (0, 255, 0), 2)\n            cv2.putText(image, labels[i], (bboxes[i][0], bboxes[i][1] - 5), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 0, 0), 2)\n\n        axes[idx].imshow(image)\n        axes[idx].axis(\"off\")\n\n    plt.show()\n    return bboxes_list  # Return updated bounding boxes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:17.374359Z","iopub.execute_input":"2025-04-08T12:09:17.374612Z","iopub.status.idle":"2025-04-08T12:09:17.383455Z","shell.execute_reply.started":"2025-04-08T12:09:17.374591Z","shell.execute_reply":"2025-04-08T12:09:17.382666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Select a random sample of unique images\nnum_samples = 5\nsampled_images = train_df[\"image_id\"].drop_duplicates().sample(n=min(num_samples, train_df[\"image_id\"].nunique())).tolist()\n\n# ✅ Prepare lists for visualization\nimage_paths, bboxes_list, labels_list, image_sizes = [], [], [], []\n\nfor image_id in sampled_images:\n    sample_df = train_df[train_df[\"image_id\"] == image_id]\n    image_path = sample_df[\"image_path\"].iloc[0]\n    print(f\"🔍 Visualizing image: {image_path}\")  # <-- ✅ Print the image path here\n    image_paths.append(image_path)\n    bboxes_list.append(sample_df[[\"x_min\", \"y_min\", \"x_max\", \"y_max\"]].values.tolist())\n    labels_list.append(sample_df[\"class_name\"].tolist())\n    image_sizes.append(sample_df[[\"width\", \"height\"]].iloc[0].tolist())\n\n# ✅ Visualize & Update Bounding Boxes\nbboxes_list_updated = visualize_bboxes(image_paths, bboxes_list, labels_list, image_sizes, num_images=num_samples)\n\n# ✅ Update train_df with new bounding boxes\nfor i, image_id in enumerate(sampled_images):\n    sample_df = train_df[train_df[\"image_id\"] == image_id].copy()\n    updated_bboxes = bboxes_list_updated[i]\n    \n    for j, bbox in enumerate(updated_bboxes):\n        train_df.loc[sample_df.index[j], ['x_min', 'y_min', 'x_max', 'y_max']] = bbox\n\nprint(\"✅ Bounding boxes updated in train_df!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:17.384218Z","iopub.execute_input":"2025-04-08T12:09:17.384416Z","iopub.status.idle":"2025-04-08T12:09:19.275181Z","shell.execute_reply.started":"2025-04-08T12:09:17.384397Z","shell.execute_reply":"2025-04-08T12:09:19.274324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Group by image_id and get unique classes per image\nunique_class_per_image = train_df.groupby(\"image_id\")[\"class_name\"].unique()\n\n# ✅ Count how many images each class appears in\nclass_counts = unique_class_per_image.explode().value_counts()\n\n# ✅ Map class_name to class_id\nclass_name_to_id = train_df.drop_duplicates(\"class_name\")[[\"class_name\", \"class_id\"]].set_index(\"class_name\")[\"class_id\"].to_dict()\n\n# ✅ Add class_id to the labels\nclass_labels_with_ids = [f\"{cls} (ID {class_name_to_id.get(cls, 'Unknown')})\" for cls in class_counts.index]\n\n# ✅ Sort class counts\nclass_counts = class_counts.sort_values(ascending=False)\nclass_labels_with_ids = [label for _, label in sorted(zip(class_counts.values, class_labels_with_ids), reverse=True)]\n\n# ✅ Create color palette\ncolors = sns.color_palette(\"tab20\", len(class_counts))\n\n# ✅ Plot the class distribution\nplt.figure(figsize=(14, 6))\nbars = plt.bar(class_labels_with_ids, class_counts.values, color=colors)\nplt.title('Class Distribution Across Images (Unique Occurrences)', fontsize=16)\nplt.xlabel('Class Name (with Class ID)', fontsize=12)\nplt.ylabel('Number of Images', fontsize=12)\nplt.xticks(rotation=45, ha='right')\n\n# ✅ Annotate counts on top of bars\nfor i, bar in enumerate(bars):\n    height = bar.get_height()\n    plt.text(bar.get_x() + bar.get_width() / 2, height + 1, str(height), ha='center', va='bottom', fontsize=10)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:19.276051Z","iopub.execute_input":"2025-04-08T12:09:19.276283Z","iopub.status.idle":"2025-04-08T12:09:19.757413Z","shell.execute_reply.started":"2025-04-08T12:09:19.276263Z","shell.execute_reply":"2025-04-08T12:09:19.756636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Group by class_id and count the occurrences of each bounding box (instead of unique classes per image)\nclass_counts = train_df['class_name'].value_counts()\n\n# ✅ Map class_name to class_id\nclass_name_to_id = train_df.drop_duplicates(\"class_name\")[[\"class_name\", \"class_id\"]].set_index(\"class_name\")[\"class_id\"].to_dict()\n\n# ✅ Add class_id to the labels\nclass_labels_with_ids = [f\"{cls} (ID {class_name_to_id.get(cls, 'Unknown')})\" for cls in class_counts.index]\n\n# ✅ Sort class counts\nclass_counts = class_counts.sort_values(ascending=False)\nclass_labels_with_ids = [label for _, label in sorted(zip(class_counts.values, class_labels_with_ids), reverse=True)]\n\n# ✅ Create color palette\ncolors = sns.color_palette(\"tab20\", len(class_counts))\n\n# ✅ Plot the class distribution\nplt.figure(figsize=(14, 6))\nbars = plt.bar(class_labels_with_ids, class_counts.values, color=colors)\nplt.title('Class Distribution Across Bounding Boxes (Total Occurrences)', fontsize=16)\nplt.xlabel('Class Name (with Class ID)', fontsize=12)\nplt.ylabel('Number of Bounding Boxes', fontsize=12)\nplt.xticks(rotation=45, ha='right')\n\n# ✅ Annotate counts on top of bars\nfor i, bar in enumerate(bars):\n    height = bar.get_height()\n    plt.text(bar.get_x() + bar.get_width() / 2, height + 1, str(height), ha='center', va='bottom', fontsize=10)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Class Mapping","metadata":{}},{"cell_type":"code","source":"# ✅ Define the class mapping dictionary without 'Fibrosis & ILD'\nclass_mapping = {\n    0: \"Cardiac & Vascular\", \n    1: \"Lung Collapse\",  \n    2: \"Cardiac & Vascular\",\n    3: \"Lung Opacity\", \n    5: \"Lung Opacity\",  \n    6: \"Lung Opacity\",  \n    7: \"Nodule/Mass or Other Lesion\", \n    8: \"Nodule/Mass or Other Lesion\",  \n    9: \"Pleural Abnormalities\",\n    10: \"Pleural Abnormalities\",  \n    11: \"Lung Collapse\",  \n    13: \"Lung Opacity\"\n}\n\n# ✅ Apply class mapping to create `mapped_class_name`\ntrain_df['mapped_class_name'] = train_df['class_id'].map(class_mapping)\n\n# ✅ Remove rows with class IDs 4 and 12 (since they are no longer mapped)\ntrain_df = train_df[~train_df['class_id'].isin([4, 12])]\n\n# ✅ Debugging: Check for unmapped class IDs\nunmapped_classes = train_df[train_df['mapped_class_name'].isna()]['class_id'].unique()\nif len(unmapped_classes) > 0:\n    print(f\"⚠️ Warning: Some class IDs are not mapped! Unmapped class IDs: {unmapped_classes}\")\n\n# ✅ Remove any NaN values before creating unique class names\ntrain_df = train_df.dropna(subset=['mapped_class_name'])\n\n# ✅ Get unique class names (ensuring correct count)\nclass_names = sorted(train_df['mapped_class_name'].unique())\n\n# ✅ Explicitly map class names to correct indices\nnew_class_ids = {name: idx for idx, name in enumerate(class_names)}\n\n# ✅ Apply the new mapping\ntrain_df['new_class_id'] = train_df['mapped_class_name'].map(new_class_ids)\n\n# ✅ Create the final new class mapping\nnew_class_mapping = {idx: name for name, idx in new_class_ids.items()}\n\n# ✅ Verify the new class mapping\nprint(f\"✅ New class mapping (new class IDs): {new_class_mapping}\")\ntrain_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:19.760346Z","iopub.execute_input":"2025-04-08T12:09:19.760612Z","iopub.status.idle":"2025-04-08T12:09:19.805608Z","shell.execute_reply.started":"2025-04-08T12:09:19.760592Z","shell.execute_reply":"2025-04-08T12:09:19.804762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Function to visualize images with bounding boxes and show image path\ndef visualize_bboxes(image_paths, bboxes_list, labels_list, image_sizes, num_images=5):\n    num_images = min(num_images, len(image_paths))  \n    fig, axes = plt.subplots(1, num_images, figsize=(50, 30))\n\n    if num_images == 1:\n        axes = [axes]\n\n    for idx in range(num_images):\n        image_path, bboxes, labels, (orig_width, orig_height) = (\n            image_paths[idx], bboxes_list[idx], labels_list[idx], image_sizes[idx])\n        \n        # ✅ Print image path to console/log\n        print(f\"\\n🖼 Visualizing image {idx + 1}/{num_images}: {image_path}\")\n        \n        image = cv2.imread(image_path)\n        if image is None:\n            print(f\"⚠️ Error: Could not read {image_path}\")\n            continue\n    \n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)  \n        height, width, _ = image.shape  \n    \n        # ✅ Rescale and draw bounding boxes\n        for i in range(len(bboxes)):\n            x_min, y_min, x_max, y_max = bboxes[i]\n            x_min = int((x_min / orig_width) * width)\n            y_min = int((y_min / orig_height) * height)\n            x_max = int((x_max / orig_width) * width)\n            y_max = int((y_max / orig_height) * height)\n            bboxes[i] = [x_min, y_min, x_max, y_max]\n            cv2.rectangle(image, (x_min, y_min), (x_max, y_max), (0, 255, 0), 2)\n            cv2.putText(image, labels[i], (x_min, y_min - 5), cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255, 0, 0), 2)\n    \n        axes[idx].imshow(image)\n        axes[idx].axis(\"off\")\n        axes[idx].set_title(image_path.split('/')[-1], fontsize=14, color='blue')  # Optional shorter title\n\n    plt.tight_layout()\n    plt.show()\n    return bboxes_list  # Return updated bounding boxes\n\n# ✅ Select a random sample of unique images\nnum_samples = 5\nsampled_images = train_df[\"image_id\"].drop_duplicates().sample(n=min(num_samples, train_df[\"image_id\"].nunique())).tolist()\n\n# ✅ Prepare lists for visualization\nimage_paths, bboxes_list, labels_list, image_sizes = [], [], [], []\n\nfor image_id in sampled_images:\n    sample_df = train_df[train_df[\"image_id\"] == image_id]\n    image_paths.append(sample_df[\"image_path\"].iloc[0])\n    bboxes_list.append(sample_df[[\"x_min\", \"y_min\", \"x_max\", \"y_max\"]].values.tolist())\n    labels_list.append(sample_df[\"mapped_class_name\"].tolist())  # Use mapped class names\n    image_sizes.append(sample_df[[\"width\", \"height\"]].iloc[0].tolist())\n\n# ✅ Visualize and update bounding boxes\nbboxes_list_updated = visualize_bboxes(image_paths, bboxes_list, labels_list, image_sizes, num_images=num_samples)\n\n# ✅ Update train_df with new bounding boxes\nfor i, image_id in enumerate(sampled_images):\n    sample_df = train_df[train_df[\"image_id\"] == image_id].copy()\n    updated_bboxes = bboxes_list_updated[i]\n    \n    for j, bbox in enumerate(updated_bboxes):\n        train_df.loc[sample_df.index[j], ['x_min', 'y_min', 'x_max', 'y_max']] = bbox\n\nprint(\"✅ Bounding boxes updated in train_df!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:19.806766Z","iopub.execute_input":"2025-04-08T12:09:19.807003Z","iopub.status.idle":"2025-04-08T12:09:22.835364Z","shell.execute_reply.started":"2025-04-08T12:09:19.806984Z","shell.execute_reply":"2025-04-08T12:09:22.834501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Group by image_id and mapped_class_name, counting unique classes per image\nunique_classes_per_image = train_df.groupby(\"image_id\")[\"mapped_class_name\"].nunique()\n\n# ✅ Count how many images have each unique class (counting each class once per image)\nclass_counts = train_df.groupby(\"mapped_class_name\")[\"image_id\"].nunique()\nclass_counts = class_counts.sort_values(ascending=False)\n\n# ✅ Create a color palette for the plot\ncolors = sns.color_palette(\"Set2\", len(class_counts))\n\n# ✅ Plot the class distribution showing how many images each class appeared in\nplt.figure(figsize=(12, 6))\nclass_counts.plot(kind='bar', color=colors)\nfor i, value in enumerate(class_counts.values):\n    plt.text(i, value + 1, str(value), ha='center', va='bottom', fontsize=10)\nplt.title('Mapped Class Distribution Across Images (Unique Occurrences)', fontsize=16)\nplt.xlabel('Mapped Class Name', fontsize=12)\nplt.ylabel('Number of Images', fontsize=12)\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()\n\n# ✅ Display the class distribution\nprint(f\"Mapped class distribution (counting each class once per image):\\n{class_counts}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:22.836339Z","iopub.execute_input":"2025-04-08T12:09:22.836617Z","iopub.status.idle":"2025-04-08T12:09:23.117434Z","shell.execute_reply.started":"2025-04-08T12:09:22.836595Z","shell.execute_reply":"2025-04-08T12:09:23.116598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Count how many times each class appears in each image (including duplicates for multiple bboxes)\nclass_occurrences = train_df.groupby([\"image_id\", \"mapped_class_name\"]).size().reset_index(name=\"count\")\n\n# ✅ Sum the counts of each class across all images\nclass_counts = class_occurrences.groupby(\"mapped_class_name\")[\"count\"].sum()\n\n# ✅ Sort by count (optional, for better visualization)\nclass_counts = class_counts.sort_values(ascending=False)\n\n# ✅ Create a color palette\ncolors = sns.color_palette(\"Set2\", len(class_counts))\n\n# ✅ Plot the class distribution with annotations\nplt.figure(figsize=(12, 6))\nbarplot = class_counts.plot(kind='bar', color=colors)\n\n# ✅ Add value annotations above each bar\nfor i, value in enumerate(class_counts.values):\n    plt.text(i, value + max(class_counts.values) * 0.01, str(value), ha='center', va='bottom', fontsize=10)\n\nplt.title('Mapped Class Distribution Across All Images (Total Bounding Box Occurrences)', fontsize=16)\nplt.xlabel('Mapped Class Name', fontsize=12)\nplt.ylabel('Total Number of Bounding Boxes', fontsize=12)\nplt.xticks(rotation=45, ha='right')\nplt.grid(axis='y', linestyle='--', alpha=0.5)\nplt.tight_layout()\nplt.show()\n\n# ✅ Print the full distribution as a summary\nprint(\"\\n📊 Mapped class distribution (counting total bounding box occurrences across images):\")\nprint(class_counts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:23.118453Z","iopub.execute_input":"2025-04-08T12:09:23.118727Z","iopub.status.idle":"2025-04-08T12:09:23.368173Z","shell.execute_reply.started":"2025-04-08T12:09:23.118704Z","shell.execute_reply":"2025-04-08T12:09:23.367279Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"WBF","metadata":{}},{"cell_type":"code","source":"# ✅ Apply WBF Function (with filtering for \"Fibrosis & ILD\" before WBF)\ndef apply_wbf(train_df, iou_thr=0.1, skip_box_thr=0.0001, min_box_size=0.01):\n    # Remove \"Fibrosis & ILD\" class before applying WBF\n    train_df = train_df[train_df['new_class_id'] != 4]  # Remove 'Fibrosis & ILD' class (class_id = 4)\n\n    output = []\n\n    for image_id, group in tqdm(train_df.groupby(\"image_id\"), desc=\"Applying WBF\"):\n        w, h = group['width'].iloc[0], group['height'].iloc[0]\n\n        boxes_list = []\n        scores_list = []\n        labels_list = []\n\n        boxes_single = []\n        labels_single = []\n\n        count_dict = Counter(group['new_class_id'].tolist())\n        class_ids = group['new_class_id'].unique().tolist()\n\n        for cid in class_ids:\n            class_group = group[group.new_class_id == cid]\n\n            if count_dict[cid] == 1:\n                row = class_group.iloc[0]\n                # Use YOLO-normalized box and convert to (x_min, y_min, x_max, y_max)\n                x_mid, y_mid, box_w, box_h = row['x_mid'], row['y_mid'], row['w'], row['h']\n                x_min = x_mid - box_w / 2\n                y_min = y_mid - box_h / 2\n                x_max = x_mid + box_w / 2\n                y_max = y_mid + box_h / 2\n\n                box = [x_min, y_min, x_max, y_max]\n                boxes_single.append(box)\n                labels_single.append(cid)\n            else:\n                # Same as above, use YOLO-normalized coords\n                x_mid = class_group['x_mid'].to_numpy()\n                y_mid = class_group['y_mid'].to_numpy()\n                box_w = class_group['w'].to_numpy()\n                box_h = class_group['h'].to_numpy()\n\n                x_min = x_mid - box_w / 2\n                y_min = y_mid - box_h / 2\n                x_max = x_mid + box_w / 2\n                y_max = y_mid + box_h / 2\n\n                bboxes = np.stack([x_min, y_min, x_max, y_max], axis=1)\n                bboxes = np.clip(bboxes, 0, 1)  # Ensure in [0,1]\n\n                boxes_list.append(bboxes.tolist())\n                scores_list.append([1.0] * len(class_group))\n                labels_list.append([cid] * len(class_group))\n\n        # Apply WBF\n        if boxes_list:\n            fused_boxes, _, fused_labels = weighted_boxes_fusion(\n                boxes_list, scores_list, labels_list,\n                weights=None, iou_thr=iou_thr, skip_box_thr=skip_box_thr\n            )\n        else:\n            fused_boxes, fused_labels = np.empty((0, 4)), np.empty((0,))\n\n        # Combine with singles\n        if len(boxes_single) > 0:\n            all_boxes = np.vstack([fused_boxes, boxes_single])\n            all_labels = np.hstack([fused_labels, labels_single])\n        else:\n            all_boxes = fused_boxes\n            all_labels = fused_labels\n\n        # Convert back to YOLO format and append\n        for box, label in zip(all_boxes, all_labels):\n            x_min, y_min, x_max, y_max = box\n            box_w = x_max - x_min\n            box_h = y_max - y_min\n            x_center = (x_min + x_max) / 2\n            y_center = (y_min + y_max) / 2\n\n            # Filter out tiny boxes\n            if box_w > min_box_size and box_h > min_box_size:\n                output.append({\n                    \"image_id\": image_id,\n                    \"x_mid\": x_center,\n                    \"y_mid\": y_center,\n                    \"w\": box_w,\n                    \"h\": box_h,\n                    \"new_class_id\": int(label) if isinstance(label, (int, float)) else label\n                })\n\n    return pd.DataFrame(output)\n\n# ✅ Display summary BEFORE WBF\nprint(\"📊 Before WBF:\")\nprint(f\"🔹 Total images: {train_df['image_id'].nunique()}\")\nprint(f\"🔹 Total labels: {len(train_df)}\\n\")\n\n# ✅ Apply WBF once (now without \"Fibrosis & ILD\")\ntrain_df_wbf = apply_wbf(train_df)\n\n# ✅ Merge additional image metadata\ntrain_df_wbf = train_df_wbf.merge(\n    train_df[['image_id', 'image_path', 'width', 'height', 'source_dataset']].drop_duplicates(),\n    on='image_id',\n    how='left'\n)\n\n# ✅ Display summary AFTER WBF\nprint(\"📊 After WBF:\")\nprint(f\"✅ Total images: {train_df_wbf['image_id'].nunique()}\")\nprint(f\"✅ Total labels: {len(train_df_wbf)}\")\n\nprint(\"\\n🔍 Sample of processed DataFrame:\")\nprint(train_df_wbf.head(5))\n\nprint(\"\\n📌 Columns in WBF output:\")\nprint(train_df_wbf.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:23.369192Z","iopub.execute_input":"2025-04-08T12:09:23.369482Z","iopub.status.idle":"2025-04-08T12:09:31.368854Z","shell.execute_reply.started":"2025-04-08T12:09:23.369455Z","shell.execute_reply":"2025-04-08T12:09:31.367975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Visualization function\ndef visualize_before_after(image_id, df_before, df_after, class_names=None):\n    img_path = df_before[df_before[\"image_id\"] == image_id].iloc[0][\"image_path\"]\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    h, w = img.shape[:2]\n\n    fig, axs = plt.subplots(1, 2, figsize=(18, 8))\n    titles = [\"Before WBF\", \"After WBF\"]\n    dfs = [df_before, df_after]\n\n    for i, (ax, title, df) in enumerate(zip(axs, titles, dfs)):\n        bboxes = df[df[\"image_id\"] == image_id]\n\n        ax.imshow(img)\n        ax.set_title(f\"{title}\", fontsize=16)\n\n        for _, row in bboxes.iterrows():\n            x_mid = row[\"x_mid\"] * w\n            y_mid = row[\"y_mid\"] * h\n            box_w = row[\"w\"] * w\n            box_h = row[\"h\"] * h\n\n            x_min = x_mid - box_w / 2\n            y_min = y_mid - box_h / 2\n\n            rect = patches.Rectangle(\n                (x_min, y_min),\n                box_w,\n                box_h,\n                linewidth=2,\n                edgecolor='lime',\n                facecolor='none'\n            )\n            ax.add_patch(rect)\n\n            class_id = int(row[\"new_class_id\"])\n            label = class_names[class_id] if class_names else str(class_id)\n            ax.text(\n                x_min, y_min - 5, label,\n                color='white',\n                fontsize=12,\n                bbox=dict(facecolor='green', alpha=0.6, edgecolor='none', pad=1)\n            )\n\n        ax.axis('off')\n\n    plt.tight_layout()\n    plt.show()\n\n# ✅ Define class names without 'Fibrosis & ILD'\nclass_names = [\n    \"Cardiac & Vascular\",\n    \"Lung Collapse\",\n    \"Lung Opacity\",\n    \"Nodule/Mass or Other Lesion\",\n    \"Pleural Abnormalities\"\n]\n\n# ✅ Adjust the 'new_class_id' mappings accordingly (shift the indices of the remaining classes)\n# This assumes that you have already removed \"Fibrosis & ILD\" from your dataset\n# If necessary, ensure that the `new_class_id` is being correctly re-mapped after removing \"Fibrosis & ILD\"\n\n# ✅ Pick 5 random image_ids\nsample_ids = random.sample(list(train_df_wbf[\"image_id\"].unique()), 5)\n\n# ✅ Visualize each image before and after WBF\nfor img_id in sample_ids:\n    visualize_before_after(img_id, train_df, train_df_wbf, class_names)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:31.369818Z","iopub.execute_input":"2025-04-08T12:09:31.370121Z","iopub.status.idle":"2025-04-08T12:09:34.892479Z","shell.execute_reply.started":"2025-04-08T12:09:31.370089Z","shell.execute_reply":"2025-04-08T12:09:34.89154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Visualize many images of the same class\n\n# Step 1: Choose class name or ID\ntarget_class_name = \"Nodule/Mass or Other Lesion\"  # change as needed\ntarget_class_id = class_names.index(target_class_name)\n\n# Step 2: Filter image_ids containing this class\nfiltered_ids = train_df[train_df[\"new_class_id\"] == target_class_id][\"image_id\"].unique()\n\n# Step 3: Pick N random image_ids (change number as needed)\nsample_ids = random.sample(list(filtered_ids), min(10, len(filtered_ids)))\n\n# Step 4: Visualize each image before and after WBF\nfor img_id in sample_ids:\n    visualize_before_after(img_id, train_df, train_df_wbf, class_names)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:34.893269Z","iopub.execute_input":"2025-04-08T12:09:34.893584Z","iopub.status.idle":"2025-04-08T12:09:42.31376Z","shell.execute_reply.started":"2025-04-08T12:09:34.893556Z","shell.execute_reply":"2025-04-08T12:09:42.312861Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. ***Data Split***","metadata":{}},{"cell_type":"code","source":"# Step 3: Multi-label per image for stratification\ntrain_df_multi = train_df_wbf.groupby('image_id')['new_class_id'].agg(lambda x: list(set(x))).reset_index()\ntrain_df_wbf = train_df_wbf.merge(train_df_multi, on='image_id', suffixes=(\"\", \"_multi\"))\ntrain_df_wbf['multi_class_str'] = train_df_wbf['new_class_id_multi'].astype(str)\n\n# Step 4: Perform the stratified group split\nsgkf = StratifiedGroupKFold(n_splits=4, shuffle=True, random_state=42)\n\nfor train_idx, val_idx in sgkf.split(train_df_wbf, train_df_wbf['multi_class_str'], groups=train_df_wbf['image_id']):\n    train_df_split = train_df_wbf.iloc[train_idx].reset_index(drop=True)\n    val_df_split = train_df_wbf.iloc[val_idx].reset_index(drop=True)\n    break\n\n# ✅ Display summary\nprint(f\"✅ Train Images: {train_df_split['image_id'].nunique()}\")\nprint(f\"✅ Val Images: {val_df_split['image_id'].nunique()}\")\nprint(\"\\n📊 Source Dataset Distribution:\")\nprint(\"Train:\")\nprint(train_df_split['source_dataset'].value_counts(normalize=True))\nprint(\"Val:\")\nprint(val_df_split['source_dataset'].value_counts(normalize=True))\n\nprint(\"\\n📚 Class Distribution:\")\nprint(\"Train:\")\nprint(train_df_split['new_class_id'].value_counts(normalize=True).sort_index())\nprint(\"Val:\")\nprint(val_df_split['new_class_id'].value_counts(normalize=True).sort_index())\n\n# Step 5: Drop temp stratification columns\ntrain_df_split.drop(columns=['new_class_id_multi', 'multi_class_str'], inplace=True)\nval_df_split.drop(columns=['new_class_id_multi', 'multi_class_str'], inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:42.314668Z","iopub.execute_input":"2025-04-08T12:09:42.3149Z","iopub.status.idle":"2025-04-08T12:09:43.58319Z","shell.execute_reply.started":"2025-04-08T12:09:42.314882Z","shell.execute_reply":"2025-04-08T12:09:43.582354Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"5. Yolo Label Files","metadata":{}},{"cell_type":"code","source":"# ✅ Create necessary directories for YOLO\nos.makedirs('data/images/train', exist_ok=True)\nos.makedirs('data/images/val', exist_ok=True)\nos.makedirs('data/labels/train', exist_ok=True)\nos.makedirs('data/labels/val', exist_ok=True)\n\n# ✅ Function to prepare YOLO labels and move the images to appropriate directories\ndef prepare_yolo_labels(df, image_dest_dir, label_dest_dir):\n    # ✅ Remove duplicate bounding boxes before writing\n    df = df.drop_duplicates(subset=['image_id', 'new_class_id', 'x_mid', 'y_mid', 'w', 'h'])\n\n    for image_id, group in tqdm(df.groupby('image_id'), desc=f\"Processing {image_dest_dir}\"):\n        image_path = group.iloc[0]['image_path']\n        label_file = os.path.join(label_dest_dir, os.path.basename(image_path).replace('.png', '.txt'))\n\n        # ✅ Copy image\n        shutil.copy(image_path, os.path.join(image_dest_dir, os.path.basename(image_path)))\n\n        # ✅ Write all labels in one go\n        with open(label_file, 'w') as f:\n            for _, row in group.iterrows():\n                f.write(f\"{row['new_class_id']} {row['x_mid']} {row['y_mid']} {row['w']} {row['h']}\\n\")\n\n\n# ✅ Prepare and move train and validation data (images and labels) with progress bars\nprepare_yolo_labels(train_df_split, 'data/images/train', 'data/labels/train')\nprepare_yolo_labels(val_df_split, 'data/images/val', 'data/labels/val')\n\n# ✅ Function to remove exact duplicate lines from YOLO label files\ndef deduplicate_yolo_labels(label_dir):\n    label_paths = glob(os.path.join(label_dir, \"*.txt\"))\n    total_files = len(label_paths)\n    deduplicated_count = 0\n    duplicate_files = []\n\n    for path in label_paths:\n        with open(path, 'r') as f:\n            lines = f.readlines()\n        original_count = len(lines)\n        deduped = list(set([line.strip() for line in lines]))\n        deduped_count = len(deduped)\n\n        if deduped_count < original_count:\n            deduplicated_count += 1\n            duplicate_files.append((os.path.basename(path), original_count - deduped_count))\n            with open(path, 'w') as f:\n                f.write('\\n'.join(deduped) + '\\n')\n\n    print(f\"✅ Deduplicated {deduplicated_count}/{total_files} files in: {label_dir}\")\n    if duplicate_files:\n        print(\"🔍 Files with duplicates removed:\")\n        for fname, dup_count in duplicate_files:\n            print(f\"{fname}: {dup_count} duplicates removed\")\n\n# ✅ Deduplicate labels for both train and val\ndeduplicate_yolo_labels('data/labels/train')\ndeduplicate_yolo_labels('data/labels/val')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:09:43.584047Z","iopub.execute_input":"2025-04-08T12:09:43.58426Z","iopub.status.idle":"2025-04-08T12:10:56.054061Z","shell.execute_reply.started":"2025-04-08T12:09:43.584243Z","shell.execute_reply":"2025-04-08T12:10:56.053011Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. ***data.yaml file & YOLO12 Training***","metadata":{}},{"cell_type":"code","source":"# ✅ Create the data.yaml file for YOLOv11\ndata_yaml = \"\"\"  \ntrain: /kaggle/working/data/images/train  \nval: /kaggle/working/data/images/val  \n\nnc: 6 \nnames: [  \n  \"Cardiac & Vascular\",  \n  \"Lung Collapse\",  \n  \"Lung Opacity\",  \n  \"Fibrosis & ILD\",  \n  \"Nodule/Mass or Other Lesion\",  \n  \"Pleural Abnormalities\"  \n]\n\"\"\"\n\n# ✅ Save the YAML file to the working directory\nwith open('/kaggle/working/data.yaml', 'w') as f:  \n    f.write(data_yaml)  \n\nprint(\"✅ data.yaml file has been created!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:10:56.05495Z","iopub.execute_input":"2025-04-08T12:10:56.055202Z","iopub.status.idle":"2025-04-08T12:10:56.060066Z","shell.execute_reply.started":"2025-04-08T12:10:56.055181Z","shell.execute_reply":"2025-04-08T12:10:56.059378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = \"/usr/local/lib/python3.10/dist-packages/ultralytics/engine/trainer.py\"\n\nwith open(path, \"r\") as f:\n    lines = f.readlines()\n\n# View a section to locate where to insert the cache-clear line\n# Check for the beginning of the training loop\nfor i, line in enumerate(lines[340:400], start=340):\n    print(f\"{i}: {line.strip()}\")\n\n# Insert cache-clear code at the appropriate lines\nlines.insert(345, \"torch.cuda.empty_cache()  # Clear CUDA cache after zeroing gradients\\n\")\nlines.insert(367, \"torch.cuda.empty_cache()  # Clear CUDA cache before processing each batch\\n\")\n\n# Save the changes back to the file\nwith open(path, \"w\") as f:\n    f.writelines(lines)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:10:56.060822Z","iopub.execute_input":"2025-04-08T12:10:56.061118Z","iopub.status.idle":"2025-04-08T12:10:56.085313Z","shell.execute_reply.started":"2025-04-08T12:10:56.061089Z","shell.execute_reply":"2025-04-08T12:10:56.084521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\ntorch.cuda.empty_cache()  # Clears unused memory in cache\n\n# ✅ Check available GPUs\nnum_gpus = torch.cuda.device_count()\nprint(f\"Available GPUs: {num_gpus}\")\nfor i in range(num_gpus):\n    print(f\"GPU {i}: {torch.cuda.get_device_name(i)}\")\n\n# ✅ Set device for training (use the first GPU if available, otherwise use CPU)\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(f\"Using {device} for training\")\n\n# ✅ Load YOLOv12-M model\nmodel = YOLO(\"yolo12m.pt\")  # Load pre-trained YOLOv12-M weights\n\n# ✅ Move model to the appropriate device (single GPU or CPU)\nmodel = model.to(device)\n\n# ✅ Optimize CUDA memory allocation\ntorch.cuda.empty_cache()\nos.environ[\"PYTORCH_CUDA_ALLOC_CONF\"] = \"expandable_segments:True\"\n\n# ✅ Train the model using the correct training method (from YOLOv12 docs)\ntrain_results = model.train(\n    data=\"/kaggle/working/data.yaml\",  # Path to dataset YAML\n    epochs=120,\n    batch=16,  \n    device=device,\n    half=True,\n    workers=2, \n    project=\"yolov12-training\",\n    name=\"yolo12m-vinbigdata\",\n    exist_ok=True,\n    save=True,\n    save_period=10,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T12:10:56.086068Z","iopub.execute_input":"2025-04-08T12:10:56.086284Z","execution_failed":"2025-04-08T19:16:09.472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Check training results\nprint(\"✅ Training results:\")\nprint(train_results)\n\n# ✅ Validate the model\nmetrics = model.val()\n\n# ✅ Print evaluation metrics\nprint(\"Evaluation metrics:\")\nprint(metrics)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T22:52:14.872503Z","iopub.execute_input":"2025-04-08T22:52:14.872816Z","iopub.status.idle":"2025-04-08T22:52:14.957056Z","shell.execute_reply.started":"2025-04-08T22:52:14.872792Z","shell.execute_reply":"2025-04-08T22:52:14.955707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!zip -r yolov12-training.zip /kaggle/working/yolov12-training/yolo12m-vinbigdata/","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Provide a clickable download link\nFileLink(r'yolov12-training.zip')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-08T10:55:56.488359Z","iopub.execute_input":"2025-04-08T10:55:56.488686Z","iopub.status.idle":"2025-04-08T10:55:56.500034Z","shell.execute_reply.started":"2025-04-08T10:55:56.488657Z","shell.execute_reply":"2025-04-08T10:55:56.498948Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ***7. Mode & 90th percentile***","metadata":{}},{"cell_type":"code","source":"!pip install ultralytics\nfrom ultralytics import YOLO","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-05T23:30:52.278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_path = \"/kaggle/input/mymodel/pytorch/yolo12/1/best.pt\"\nmodel = YOLO(model_path)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-05T23:30:52.278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\nimport glob\nfrom collections import defaultdict, Counter\n\n# Move model to GPU\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.to(device)\n\n# Get all image file paths\nimage_dir = \"/kaggle/input/vinbigdata-1024-image-dataset/vinbigdata/test/\"\nimage_paths = sorted(glob.glob(image_dir + \"*.png\"))  # Adjust extension if needed\n\nbatch_size = 16  # Adjust based on GPU memory\nall_results = []\n\n# Run inference in batches\nfor i in range(0, len(image_paths), batch_size):\n    batch = image_paths[i : i + batch_size]  # Get batch of image paths\n    batch_results = model(batch, conf=0.1)  # Run inference\n    all_results.extend(batch_results)\n\n# Collect confidence scores by class\nconf_scores = defaultdict(list)\n\nfor result in all_results:\n    if result.boxes is not None:  # Ensure detections exist\n        for det in result.boxes.to(device):  # Keep tensors on GPU\n            cls = int(det.cls.item())  # Get class ID\n            conf = float(det.conf.item())  # Get confidence score\n            class_name = model.names[cls] if hasattr(model, \"names\") else str(cls)\n            conf_scores[class_name].append(conf)\n\n# Calculate the mode (most common confidence score) and 90th percentile for each class\nclass_conf_stats = {}\n\nfor cls, scores in conf_scores.items():\n    # Calculate the mode\n    mode_conf = Counter(scores).most_common(1)[0][0]\n    \n    # Calculate the 90th percentile\n    percentile_90 = np.percentile(scores, 90)\n    \n    class_conf_stats[cls] = {\n        \"mode\": mode_conf,\n        \"90th_percentile\": percentile_90\n    }\n\nprint(class_conf_stats)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-05T23:30:52.278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# modes & 90th percentile for each class\nclass_conf_data = {\n    'Cardiac & Vascular': {'mode': 0.131, '90th_percentile': 0.705},\n    'Pleural Abnormalities': {'mode': 0.115, '90th_percentile': 0.389},\n    'Lung Collapse': {'mode': 0.111, '90th_percentile': 0.507},\n    'Fibrosis & ILD': {'mode': 0.337, '90th_percentile': 0.635},\n    'Nodule/Mass or Other Lesion': {'mode': 0.208, '90th_percentile': 0.524},\n    'Lung Opacity': {'mode': 0.195, '90th_percentile': 0.515}\n}\n\nadjusted_thresholds = {}\n\nfor cls, values in class_conf_data.items():\n    mode = values['mode']\n    perc90 = values['90th_percentile']\n\n    if mode < 0.2:\n        # Use the 75th percentile if mode is too low\n        new_threshold = np.percentile([mode, perc90], 75)\n    elif mode >= 0.3:\n        # Use the mode directly if reasonable\n        new_threshold = mode\n    else:\n        # Use a weighted average if 90th percentile is much higher\n        new_threshold = (0.7 * mode) + (0.3 * perc90)\n\n    adjusted_thresholds[cls] = round(new_threshold, 3)\n\nprint(adjusted_thresholds)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-05T23:30:52.278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nimport torch\nfrom PIL import Image\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\nfrom ultralytics import YOLO  \n\n# ✅ Force inline display in Kaggle\n%matplotlib inline  \n\n# Load YOLO model\nmodel_path = \"/kaggle/input/mymodel/pytorch/yolo12/1/best.pt\"\nmodel = YOLO(model_path)\n\n# Confidence thresholds\nadjusted_thresholds = {\n    'Cardiac & Vascular': 0.562,\n    'Pleural Abnormalities': 0.32,\n    'Lung Collapse': 0.408,\n    'Fibrosis & ILD': 0.337,\n    'Nodule/Mass or Other Lesion': 0.303,\n    'Lung Opacity': 0.435\n}\n\n# Unique colors per class\nclass_colors = {\n    'Cardiac & Vascular': (255, 0, 0),\n    'Pleural Abnormalities': (0, 255, 0),\n    'Lung Collapse': (0, 0, 255),\n    'Fibrosis & ILD': (255, 255, 0),\n    'Nodule/Mass or Other Lesion': (255, 165, 0),\n    'Lung Opacity': (128, 0, 128)\n}\n\n# Dataset path\ndataset_path = Path(\"/kaggle/input/testing/\")\n\n# Function to preprocess image\ndef preprocess_image(image_path):\n    img = Image.open(image_path).convert(\"RGB\")\n    return img\n\n# Process images\nfor image_path in dataset_path.glob(\"*.*\"):\n    if image_path.suffix.lower() in [\".png\", \".jpg\", \".jpeg\"]:\n        processed_img = preprocess_image(image_path)\n        img_array = np.array(processed_img)  \n        img_height, img_width = img_array.shape[:2]  \n\n        # Run inference\n        results = model(processed_img)[0]  \n\n        # Extract boxes, confidences, and classes\n        boxes = results.boxes.xyxy.cpu().numpy()\n        confidences = results.boxes.conf.cpu().numpy()\n        classes = results.boxes.cls.cpu().numpy().astype(int)\n\n        filtered_boxes, filtered_classes, filtered_confidences = [], [], []\n\n        for i in range(len(classes)):\n            class_idx = classes[i]\n            class_name = model.names[class_idx]\n            conf = confidences[i]\n\n            class_threshold = adjusted_thresholds.get(class_name, 0.1)\n            if conf >= class_threshold:\n                filtered_boxes.append(boxes[i])\n                filtered_classes.append(class_name)\n                filtered_confidences.append(conf)\n\n        # If valid detections exist\n        if filtered_boxes:\n            image = np.array(processed_img)  \n\n            for i in range(len(filtered_boxes)):\n                x1, y1, x2, y2 = map(int, filtered_boxes[i])\n\n                # Assign unique color\n                color = class_colors.get(filtered_classes[i], (255, 255, 255))  \n\n                # Dynamic font scaling\n                font_scale = max(0.5, min(img_width, img_height) / 600)  # Adjusted for image size\n                thickness = max(2, int(font_scale * 2))  # Bold text by increasing thickness\n\n                # Draw bounding box\n                cv2.rectangle(image, (x1, y1), (x2, y2), color, thickness)  \n\n                # Text label (No background)\n                label = f\"{filtered_classes[i]} ({filtered_confidences[i]:.2f})\"\n                \n                # Draw text\n                cv2.putText(image, label, (x1, y1 - 5), cv2.FONT_HERSHEY_SIMPLEX, font_scale, color, thickness)\n\n            # ✅ Convert BGR to RGB for proper display\n            image_rgb = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n            # ✅ Ensure images appear properly\n            plt.figure(figsize=(8, 8))\n            plt.imshow(image_rgb)  # Show the image in RGB format\n            plt.axis(\"off\")\n            plt.title(f\"Detections for {image_path.name}\")\n            plt.show()\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-05T23:30:52.278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%matplotlib inline\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-04-05T23:30:52.278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-06T23:04:19.35857Z","iopub.execute_input":"2025-04-06T23:04:19.358977Z","iopub.status.idle":"2025-04-06T23:04:19.365245Z","shell.execute_reply.started":"2025-04-06T23:04:19.358945Z","shell.execute_reply":"2025-04-06T23:04:19.364387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}