{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":1810938,"sourceType":"datasetVersion","datasetId":1075803},{"sourceId":2062539,"sourceType":"datasetVersion","datasetId":1236259}],"dockerImageVersionId":31154,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nprint(\"CUDA available?\", torch.cuda.is_available())\nprint(\"GPU device:\", torch.cuda.get_device_name(0) if torch.cuda.is_available() else \"CPU\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:04:24.744037Z","iopub.execute_input":"2025-10-27T18:04:24.744292Z","iopub.status.idle":"2025-10-27T18:04:26.601458Z","shell.execute_reply.started":"2025-10-27T18:04:24.744273Z","shell.execute_reply":"2025-10-27T18:04:26.600533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install numpy<2.0 --force-reinstall\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport cv2\nimport os\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:04:26.603264Z","iopub.execute_input":"2025-10-27T18:04:26.603612Z","iopub.status.idle":"2025-10-27T18:04:27.170149Z","shell.execute_reply.started":"2025-10-27T18:04:26.603589Z","shell.execute_reply":"2025-10-27T18:04:27.167507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('../input/vinbigdata-chest-xray-resized-png-1024x1024/train_meta.csv')\nprint(df.head())\n\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-10-28T05:44:56.749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:04:27.242608Z","iopub.execute_input":"2025-10-27T18:04:27.242877Z","iopub.status.idle":"2025-10-27T18:04:33.803913Z","shell.execute_reply.started":"2025-10-27T18:04:27.24285Z","shell.execute_reply":"2025-10-27T18:04:33.802846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"annotations = pd.read_csv('/kaggle/input/vinbigdatachestxrayabnormalitiesdetection/train.csv')\ntrain_meta = pd.read_csv('/kaggle/input/vinbigdata-chest-xray-resized-png-1024x1024/train_meta.csv')\nprint(annotations.columns)\nprint(annotations.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:04:33.804946Z","iopub.execute_input":"2025-10-27T18:04:33.805174Z","iopub.status.idle":"2025-10-27T18:04:33.928045Z","shell.execute_reply.started":"2025-10-27T18:04:33.805156Z","shell.execute_reply":"2025-10-27T18:04:33.927255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"annotations = pd.read_csv('/kaggle/input/vinbigdatachestxrayabnormalitiesdetection/train.csv')\nprint(annotations.columns)\nprint(annotations.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:04:33.930341Z","iopub.execute_input":"2025-10-27T18:04:33.930672Z","iopub.status.idle":"2025-10-27T18:04:34.021522Z","shell.execute_reply.started":"2025-10-27T18:04:33.930654Z","shell.execute_reply":"2025-10-27T18:04:34.020621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\ndef plot_image_with_boxes(img_id, annotations, img_dir):\n    img_path = f\"{img_dir}/{img_id}.png\"\n    image = cv2.imread(img_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    \n    # Get all findings (not 'No finding') for the image\n    rows = annotations[(annotations['image_id'] == img_id) & (annotations['class_name'] != 'No finding')]\n    for idx, row in rows.iterrows():\n        x1, y1, x2, y2 = int(row['x_min']), int(row['y_min']), int(row['x_max']), int(row['y_max'])\n        label = row['class_name']\n        cv2.rectangle(image, (x1, y1), (x2, y2), (255,0,0), 2)\n        cv2.putText(image, label, (x1, y1-10), cv2.FONT_HERSHEY_SIMPLEX, 0.7, (255,0,0), 2)\n    \n    plt.figure(figsize=(10,10))\n    plt.imshow(image)\n    plt.axis('off')\n    plt.show()\n\n# Example usage\nplot_image_with_boxes('1c32170b4af4ce1a3030eb8167753b06', annotations, '/kaggle/input/vinbigdata-chest-xray-resized-png-1024x1024/train')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:04:34.022452Z","iopub.execute_input":"2025-10-27T18:04:34.022857Z","iopub.status.idle":"2025-10-27T18:04:34.524605Z","shell.execute_reply.started":"2025-10-27T18:04:34.022826Z","shell.execute_reply":"2025-10-27T18:04:34.523855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#List all classes from your DataFrame\nprint(sorted(annotations['class_name'].unique()))\nprint('Total classes:', len(annotations['class_name'].unique()))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:04:34.525431Z","iopub.execute_input":"2025-10-27T18:04:34.525705Z","iopub.status.idle":"2025-10-27T18:04:34.538675Z","shell.execute_reply.started":"2025-10-27T18:04:34.525665Z","shell.execute_reply":"2025-10-27T18:04:34.53799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Example PyTorch Classes Encoding\nfrom sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\nannotations['class_idx'] = label_encoder.fit_transform(annotations['class_name'])\nprint(label_encoder.classes_)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:04:34.539582Z","iopub.execute_input":"2025-10-27T18:04:34.540216Z","iopub.status.idle":"2025-10-27T18:04:35.016222Z","shell.execute_reply.started":"2025-10-27T18:04:34.540194Z","shell.execute_reply":"2025-10-27T18:04:35.015478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\n\n# Paths (adjust as necessary)\nIMG_DIR = '/kaggle/input/vinbigdata-chest-xray-resized-png-1024x1024/train'\nYOLO_LABEL_DIR = '/kaggle/working/labels_yolo'  # Output dir for YOLO txt files\n\nos.makedirs(YOLO_LABEL_DIR, exist_ok=True)\n\n# Get image dimensions from the metadata file if needed\nmeta_df = pd.read_csv('/kaggle/input/vinbigdata-chest-xray-resized-png-1024x1024/train_meta.csv')\nimg_dim_dict = dict(zip(meta_df['image_id'], zip(meta_df['dim1'], meta_df['dim0'])))\n\nfor img_id, group in annotations.groupby('image_id'):\n    label_lines = []\n    for _, row in group.iterrows():\n        if row['class_name'] == 'No finding':  # Skip if no finding\n            continue\n        # Get original image width and height\n        img_w, img_h = img_dim_dict[img_id]\n        \n        # YOLO format: <class_id> <x_center_norm> <y_center_norm> <w_norm> <h_norm>\n        x_min, y_min, x_max, y_max = row['x_min'], row['y_min'], row['x_max'], row['y_max']\n        # Normalize\n        x_center = ((x_min + x_max) / 2) / img_w\n        y_center = ((y_min + y_max) / 2) / img_h\n        w = (x_max - x_min) / img_w\n        h = (y_max - y_min) / img_h\n        # YOLO class id (already encoded previously)\n        class_idx = int(row['class_idx'])\n        label_line = f\"{class_idx} {x_center:.6f} {y_center:.6f} {w:.6f} {h:.6f}\"\n        label_lines.append(label_line)\n        \n    # Write labels only if there are findings\n    if label_lines:\n        with open(f\"{YOLO_LABEL_DIR}/{img_id}.txt\", \"w\") as f:\n            for l in label_lines:\n                f.write(l + \"\\n\")\n\nprint(\"YOLO label file conversion completed.\")\nprint(f\"Sample txt: {YOLO_LABEL_DIR}/{img_id}.txt\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:04:35.01701Z","iopub.execute_input":"2025-10-27T18:04:35.017382Z","iopub.status.idle":"2025-10-27T18:04:40.105858Z","shell.execute_reply.started":"2025-10-27T18:04:35.017356Z","shell.execute_reply":"2025-10-27T18:04:40.105068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:04:40.106883Z","iopub.execute_input":"2025-10-27T18:04:40.107298Z","iopub.status.idle":"2025-10-27T18:04:46.836707Z","shell.execute_reply.started":"2025-10-27T18:04:40.107268Z","shell.execute_reply":"2025-10-27T18:04:46.83585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nimport random\nfrom glob import glob\n\nIMG_SRC = IMG_DIR  # Your VinBigData images folder\nTXT_SRC = YOLO_LABEL_DIR  # Your YOLO labels\nTRAIN_IMG_DST = '/kaggle/working/dataset/images/train/'\nVAL_IMG_DST = '/kaggle/working/dataset/images/val/'\nTRAIN_LABEL_DST = '/kaggle/working/dataset/labels/train/'\nVAL_LABEL_DST = '/kaggle/working/dataset/labels/val/'\n\nfor d in [TRAIN_IMG_DST, VAL_IMG_DST, TRAIN_LABEL_DST, VAL_LABEL_DST]:\n    os.makedirs(d, exist_ok=True)\n\nall_imgs = glob(f\"{IMG_SRC}/*.png\")\nrandom.shuffle(all_imgs)\nsplit_idx = int(0.8 * len(all_imgs))  # 80/20 split\ntrain_imgs, val_imgs = all_imgs[:split_idx], all_imgs[split_idx:]\n\n# Move train images and labels\nfor img_path in train_imgs:\n    img_id = os.path.splitext(os.path.basename(img_path))[0]\n    shutil.copy(img_path, TRAIN_IMG_DST)\n    label_path = f\"{TXT_SRC}/{img_id}.txt\"\n    if os.path.exists(label_path):\n        shutil.copy(label_path, TRAIN_LABEL_DST)\n# Move val images and labels\nfor img_path in val_imgs:\n    img_id = os.path.splitext(os.path.basename(img_path))[0]\n    shutil.copy(img_path, VAL_IMG_DST)\n    label_path = f\"{TXT_SRC}/{img_id}.txt\"\n    if os.path.exists(label_path):\n        shutil.copy(label_path, VAL_LABEL_DST)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:04:46.837798Z","iopub.execute_input":"2025-10-27T18:04:46.838085Z","iopub.status.idle":"2025-10-27T18:36:01.543494Z","shell.execute_reply.started":"2025-10-27T18:04:46.838059Z","shell.execute_reply":"2025-10-27T18:36:01.542756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train images:\", len(glob('/kaggle/working/dataset/images/train/*.png')))\nprint(\"Train labels:\", len(glob('/kaggle/working/dataset/labels/train/*.txt')))\nprint(\"Val images:\", len(glob('/kaggle/working/dataset/images/val/*.png')))\nprint(\"Val labels:\", len(glob('/kaggle/working/dataset/labels/val/*.txt')))\nprint(\"Sample train images:\", glob('/kaggle/working/dataset/images/train/*.png')[:5])\nprint(\"Sample val labels:\", glob('/kaggle/working/dataset/labels/val/*.txt')[:5])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:36:01.544465Z","iopub.execute_input":"2025-10-27T18:36:01.545025Z","iopub.status.idle":"2025-10-27T18:36:01.606784Z","shell.execute_reply.started":"2025-10-27T18:36:01.544997Z","shell.execute_reply":"2025-10-27T18:36:01.606163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip uninstall -y numpy\n!pip install numpy==1.26.4\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-10-28T05:44:56.75Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Write the YOLO data config\ndata_yaml = \"\"\"\ntrain: /kaggle/working/dataset/images/train\nval: /kaggle/working/dataset/images/val\nnc: 15\nnames: ['Aortic enlargement', 'Atelectasis', 'Calcification', 'Cardiomegaly',\n        'Consolidation', 'ILD', 'Infiltration', 'Lung Opacity', 'No finding',\n        'Nodule/Mass', 'Other lesion', 'Pleural effusion', 'Pleural thickening',\n        'Pneumothorax', 'Pulmonary fibrosis']\n\"\"\"\nwith open('/kaggle/working/data.yaml', 'w') as f:\n    f.write(data_yaml)\n\n# Model training\nfrom ultralytics import YOLO\n\nmodel = YOLO('yolov8m.pt')  # or yolov8s.pt for small model\n# Large images, smaller batch\nresults = model.train(\n    data='/kaggle/working/data.yaml',\n    epochs=40,\n    imgsz=1280,      # Higher resolution\n    batch=4,         # Lower batch size to fit in RAM\n    device=0,\n    save_period=1\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-27T18:36:01.607413Z","iopub.execute_input":"2025-10-27T18:36:01.607651Z","execution_failed":"2025-10-28T05:44:56.747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ultralytics --upgrade --quiet\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-10-28T05:44:56.748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\nmodel = YOLO('yolov8m.pt')\nresults = model.train(\n    data='/kaggle/working/data.yaml',\n    epochs=40,\n    imgsz=1024,\n    batch=8,\n    device=0\n)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-10-28T05:44:56.749Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import os\nprint(os.listdir('/kaggle/working/runs/detect/train/'))   # adjust if path is different\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-10-28T05:44:56.749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}