{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":1799839,"datasetId":1069682,"databundleVersionId":1837296},{"sourceType":"datasetVersion","sourceId":12853590,"datasetId":7999273,"databundleVersionId":13491956}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import os\n# import pydicom\n# from tqdm import tqdm  # progress bar\n\n# dicom_dir = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/\"\n# sizes = set()\n\n# # loop with progress bar\n# for file in tqdm(os.listdir(dicom_dir), desc=\"Reading DICOM files\"):\n#     file_path = os.path.join(dicom_dir, file)\n#     try:\n#         dcm = pydicom.dcmread(file_path)\n#         sizes.add((dcm.Columns, dcm.Rows))  # (width, height)\n#     except:\n#         pass  # skip non-DICOM files\n\n# print(\"Unique image sizes:\", sizes)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-24T14:16:51.709694Z","iopub.execute_input":"2025-08-24T14:16:51.710246Z","iopub.status.idle":"2025-08-24T14:16:51.713494Z","shell.execute_reply.started":"2025-08-24T14:16:51.710217Z","shell.execute_reply":"2025-08-24T14:16:51.712958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n\ncsv_path = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv\"\ndf = pd.read_csv(csv_path)\n\n\ndisease_counts = df[\"class_name\"].value_counts()\n\nprint(\"Number of images per disease:\")\nprint(disease_counts)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-24T14:16:51.723642Z","iopub.execute_input":"2025-08-24T14:16:51.724281Z","iopub.status.idle":"2025-08-24T14:16:51.845583Z","shell.execute_reply.started":"2025-08-24T14:16:51.724256Z","shell.execute_reply":"2025-08-24T14:16:51.844842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n\ncsv_path = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv\"\ndf = pd.read_csv(csv_path)\n\nkeep_classes = [\"Aortic enlargement\", \"Cardiomegaly\", \"Lung Opacity\", \"Pleural effusion\"]\n\n\nfiltered_df = df[df[\"class_name\"].isin(keep_classes)]\n\n\nprint(filtered_df[\"class_name\"].value_counts())\n\nfiltered_df.to_csv(\"filtered_train.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-24T14:16:51.84682Z","iopub.execute_input":"2025-08-24T14:16:51.847135Z","iopub.status.idle":"2025-08-24T14:16:52.031222Z","shell.execute_reply.started":"2025-08-24T14:16:51.847108Z","shell.execute_reply":"2025-08-24T14:16:52.030484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndf = pd.read_csv(\"/kaggle/working/filtered_train.csv\")\ndisease_counts = df[\"class_name\"].value_counts()\n\nprint(\"Number of images per disease:\")\nprint(disease_counts)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-24T14:16:52.032008Z","iopub.execute_input":"2025-08-24T14:16:52.032236Z","iopub.status.idle":"2025-08-24T14:16:52.058404Z","shell.execute_reply.started":"2025-08-24T14:16:52.032217Z","shell.execute_reply":"2025-08-24T14:16:52.057722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pydicom\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\n\ncsv_file = \"/kaggle/working/filtered_train.csv\"\ndicom_folder = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train\"\noutput_dir = \"/kaggle/working/yolo_dataset\"\n\n\nos.makedirs(f\"{output_dir}/images/train\", exist_ok=True)\nos.makedirs(f\"{output_dir}/images/val\", exist_ok=True)\nos.makedirs(f\"{output_dir}/labels/train\", exist_ok=True)\nos.makedirs(f\"{output_dir}/labels/val\", exist_ok=True)\n\n\ndf = pd.read_csv(csv_file)\n\n\nkeep_classes = [\"Aortic enlargement\", \"Cardiomegaly\", \"Lung Opacity\", \"Pleural effusion\"]\ndf = df[df[\"class_name\"].isin(keep_classes)]\n\n\nclass_map = {cls: i for i, cls in enumerate(keep_classes)}\nprint(\"Class mapping:\", class_map)\n\n# === Train/Val Split ===\nimage_ids = df[\"image_id\"].unique()\ntrain_ids, val_ids = train_test_split(image_ids, test_size=0.1, random_state=42)\n\n\nfor img_id in tqdm(image_ids, desc=\"Converting DICOM to PNG + labels\"):\n    dcm_path = os.path.join(dicom_folder, f\"{img_id}.dicom\")  \n    if not os.path.exists(dcm_path):\n        continue\n\n   \n    dcm = pydicom.dcmread(dcm_path)\n    img = dcm.pixel_array\n\n    # Normalize → uint8\n    img = cv2.convertScaleAbs(img, alpha=(255.0 / img.max()))\n\n    \n    if img_id in train_ids:\n        img_out = f\"{output_dir}/images/train/{img_id}.png\"\n        label_out = f\"{output_dir}/labels/train/{img_id}.txt\"\n    else:\n        img_out = f\"{output_dir}/images/val/{img_id}.png\"\n        label_out = f\"{output_dir}/labels/val/{img_id}.txt\"\n\n    \n    cv2.imwrite(img_out, img)\n\n   \n    rows = df[df[\"image_id\"] == img_id]\n    h, w = img.shape\n    with open(label_out, \"w\") as f:\n        for _, row in rows.iterrows():\n            cls_id = class_map[row[\"class_name\"]]\n            # YOLO bbox: x_center, y_center, width, height (normalized)\n            x_min, y_min, x_max, y_max = row[\"x_min\"], row[\"y_min\"], row[\"x_max\"], row[\"y_max\"]\n            x_center = (x_min + x_max) / 2 / w\n            y_center = (y_min + y_max) / 2 / h\n            bw = (x_max - x_min) / w\n            bh = (y_max - y_min) / h\n            f.write(f\"{cls_id} {x_center} {y_center} {bw} {bh}\\n\")\n\nprint(\"✅ Conversion completed! PNG + YOLO labels are ready.\")\n\n\ndef count_images_per_class(subset_ids, subset_name):\n    subset_df = df[df[\"image_id\"].isin(subset_ids)]\n    counts = subset_df.groupby(\"class_name\")[\"image_id\"].nunique()\n    print(f\"\\n📊 {subset_name} set class distribution (unique images):\")\n    print(counts)\n\ncount_images_per_class(train_ids, \"Train\")\ncount_images_per_class(val_ids, \"Val\")\n\n\nassert set(train_ids).isdisjoint(set(val_ids)), \"⚠️ Data leakage detected!\"\nprint(\"\\n✅ No data leakage: train/val sets are completely separate.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-24T14:16:52.060063Z","iopub.execute_input":"2025-08-24T14:16:52.060291Z","iopub.status.idle":"2025-08-24T15:26:21.778667Z","shell.execute_reply.started":"2025-08-24T14:16:52.060256Z","shell.execute_reply":"2025-08-24T15:26:21.777812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-24T15:26:21.779451Z","iopub.execute_input":"2025-08-24T15:26:21.779713Z","iopub.status.idle":"2025-08-24T15:27:45.447598Z","shell.execute_reply.started":"2025-08-24T15:26:21.779693Z","shell.execute_reply":"2025-08-24T15:27:45.446889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import yaml\n\nclasses = [\"Aortic enlargement\", \"Cardiomegaly\", \"Lung Opacity\", \"Pleural effusion\"]\n\ndata_yaml = {\n    'train': '/kaggle/working/yolo_dataset/images/train',\n    'val': '/kaggle/working/yolo_dataset/images/val',\n    'nc': len(classes),\n    'names': classes\n}\n\nwith open(\"/kaggle/working/chest_xray.yaml\", 'w') as f:\n    yaml.dump(data_yaml, f, default_flow_style=False)\n\nprint(\"✅ chest_xray.yaml created!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-24T15:27:45.448575Z","iopub.execute_input":"2025-08-24T15:27:45.448845Z","iopub.status.idle":"2025-08-24T15:27:45.502593Z","shell.execute_reply.started":"2025-08-24T15:27:45.448808Z","shell.execute_reply":"2025-08-24T15:27:45.501938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\nmodel = YOLO(\"/kaggle/input/yolooo/yolo11m.pt\")   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-24T15:27:45.503367Z","iopub.execute_input":"2025-08-24T15:27:45.504065Z","iopub.status.idle":"2025-08-24T15:27:52.088881Z","shell.execute_reply.started":"2025-08-24T15:27:45.504039Z","shell.execute_reply":"2025-08-24T15:27:52.088066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. Train the model\nmodel.train(\n    data=\"chest_xray.yaml\",  \n    epochs=50,               \n    imgsz=640,              \n    batch=16,                \n    device=0                 \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-24T16:10:56.932561Z","iopub.execute_input":"2025-08-24T16:10:56.932855Z"}},"outputs":[],"execution_count":null}]}