{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.13"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":24800,"databundleVersionId":1831594,"sourceType":"competition"},{"sourceId":1799839,"sourceType":"datasetVersion","datasetId":1069682},{"sourceId":1801996,"sourceType":"datasetVersion","datasetId":1069999},{"sourceId":1810939,"sourceType":"datasetVersion","datasetId":1075804},{"sourceId":2160905,"sourceType":"datasetVersion","datasetId":1297065},{"sourceId":13149551,"sourceType":"datasetVersion","datasetId":8331267},{"sourceId":13281014,"sourceType":"datasetVersion","datasetId":8416862},{"sourceId":13785408,"sourceType":"datasetVersion","datasetId":8775602},{"sourceId":14074620,"sourceType":"datasetVersion","datasetId":8858697},{"sourceId":14084988,"sourceType":"datasetVersion","datasetId":8967574}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:24:07.15151Z","iopub.execute_input":"2026-01-08T14:24:07.151805Z","iopub.status.idle":"2026-01-08T14:25:21.646585Z","shell.execute_reply.started":"2026-01-08T14:24:07.151777Z","shell.execute_reply":"2026-01-08T14:25:21.645636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport os\nimport torch.nn\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom typing import Callable, Any\ntqdm.pandas()\n\nimport cv2\nimport shutil\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport pydicom\nimport yaml\nimport glob\nfrom PIL import Image \nfrom ultralytics import YOLO\nfrom PIL import Image\nfrom torch import nn, optim\nfrom torchvision import transforms, utils\nfrom torch.utils.data import DataLoader, Dataset\nfrom sklearn.preprocessing import MultiLabelBinarizer \nfrom sklearn.model_selection import train_test_split\nfrom skmultilearn.model_selection import iterative_train_test_split\nfrom skimage import exposure","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:25:21.648138Z","iopub.execute_input":"2026-01-08T14:25:21.648387Z","iopub.status.idle":"2026-01-08T14:25:29.351891Z","shell.execute_reply.started":"2026-01-08T14:25:21.648361Z","shell.execute_reply":"2026-01-08T14:25:29.351124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\ndevice","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:25:29.352683Z","iopub.execute_input":"2026-01-08T14:25:29.353053Z","iopub.status.idle":"2026-01-08T14:25:29.439496Z","shell.execute_reply.started":"2026-01-08T14:25:29.353023Z","shell.execute_reply":"2026-01-08T14:25:29.438581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ***Utils***\n\nimport multiprocessing\nfrom joblib import Parallel, delayed\nimport pydicom\nimport pydicom\nimport numpy as np\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef dicom2array(path, voi_lut=True, fix_monochrome=True):\n    # Đọc file DICOM\n    dicom = pydicom.dcmread(path)\n\n    # Áp dụng VOI LUT (windowing)\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array.astype(np.float32)\n\n    # MONOCHROME1 cần đảo ngược pixel\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.max(data) - data\n\n    # Chuẩn hóa về 0–1\n    data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n\n    # Scale về 0–255\n    data = (data * 255).astype(np.uint8)\n\n    return data\n\ndef resize_boxes(row, h_resize, w_resize):\n    if pd.isna(row['x_min']):\n        return pd.Series([row['x_min'], row['y_min'], row['x_max'],  row['y_max']])\n\n    w_scale = w_resize/ row['width']\n    h_scale = h_resize/ row['height']\n\n    x_min_new = round(row['x_min'] * w_scale, 1)\n    x_max_new = round(row['x_max'] * w_scale, 1)\n    y_min_new = round(row['y_min'] * h_scale, 1)\n    y_max_new = round(row['y_max'] * h_scale, 1)\n\n    return pd.Series([x_min_new, y_min_new, x_max_new, y_max_new])\n\n# Hàm lấy tỉ lệ tương đối của bboxes\ndef normalize_bbox(df):\n    df['x_center'] = (df['x_min'] + df['x_max'])/2\n    df['y_center'] = (df['y_min'] + df['y_max'])/2\n    df['bbox_width'] = df['x_max'] - df['x_min']\n    df['bbox_height'] = df['y_max'] - df['y_min']\n\n    df['x_center_norm'] = df['x_center'] / df['width']\n    df['y_center_norm'] = df['y_center'] / df['height']\n    df['bbox_height_norm'] = df['bbox_height'] / df['height']\n    df['bbox_width_norm'] = df['bbox_width'] / df['width']\n    return df\n\n# Hàm ghi thông \ndef get_bbox(df, output_file):\n    with open(output_file, 'w') as f:\n        for _, row in df.iterrows():\n            class_id = row['class_id']\n            x_center, y_center = row['x_center_norm'], row['y_center_norm']\n            width, height = row[\"bbox_width_norm\"], row['bbox_height_norm']\n            f.write(f\"{class_id} {x_center} {y_center} {width} {height}\\n\")\n            \ndef plot_image_with_bounding_box(image, bounding_boxes, class_dict):\n    fig, ax = plt.subplots()\n    ax.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n    \n    for box in bounding_boxes:\n        class_id, x, y, width, height = map(float, box.split())\n        image_width, image_height = image.shape[1], image.shape[0]\n        x1 = int((x - width / 2) * image_width)\n        y1 = int((y - height / 2) * image_height)\n        x2 = int((x + width / 2) * image_width)\n        y2 = int((y + height / 2) * image_height)\n        \n        # Choose random color for bounding box\n        color = [random.random() for _ in range(3)]\n        \n        rect = plt.Rectangle((x1, y1), x2 - x1, y2 - y1, linewidth=2, edgecolor=color, facecolor='none')\n        ax.add_patch(rect)\n        \n        # Add label text using class label from dictionary\n        label_text = class_dict[int(class_id)]\n        ax.text(x1, y1, label_text, color='white', verticalalignment='top', bbox={'color': color, 'pad': 0})\n    \n    plt.show()\n\ndef main(image_path, bounding_box_path, class_dict):\n    # Read image\n    print(type(image_path))\n    image = cv2.imread(image_path)\n    \n    # Read bounding boxes\n    with open(bounding_box_path, 'r') as file:\n        bounding_boxes = file.readlines()\n    \n    # Plot image with bounding boxes\n    plot_image_with_bounding_box(image, bounding_boxes, class_dict)\n    \n# Hàm plot hình ảnh\ndef plot_img(imgs, cols=4, size=7, title=\"\", cmap='gray', is_rgb=True, img_size=(500, 500)):\n    rows = len(imgs) // cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows, cols, i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n\n# Hàm cân bằng histogram cho ảnh\ndef hist_equalize(img_path_with_output):\n    img_path, output_dir = img_path_with_output\n    filename = os.path.splitext(os.path.basename(img_path))[0]\n    img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)\n    equalize_img = exposure.equalize_hist(img)\n    equalize_img = (equalize_img * 255).astype(np.uint8)\n    cv2.imwrite(os.path.join(output_dir, f'{filename}.png'), equalize_img)\n\n# Hàm lưu \ndef save_img(img_path_list, output_dir, n_jobs = -1):\n    os.makedirs(output_dir, exist_ok=True)\n    img_output_list = [(path, output_dir) for path in img_path_list]\n\n    Parallel(n_jobs = n_jobs)(\n        delayed(hist_equalize)(args) for args in tqdm(img_output_list, desc=\"Histogram Equalizing\")\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:25:29.440939Z","iopub.execute_input":"2026-01-08T14:25:29.441239Z","iopub.status.idle":"2026-01-08T14:25:31.926075Z","shell.execute_reply.started":"2026-01-08T14:25:29.441221Z","shell.execute_reply":"2026-01-08T14:25:31.925457Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ***1. Phân tích data***","metadata":{}},{"cell_type":"markdown","source":"## **Loading data**","metadata":{}},{"cell_type":"code","source":"sourse_path = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/\"\nbboxes_path = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv\"\ndicom_path = \"/kaggle/input/vinbigdata/train_meta.csv\"\n\ns_csv = pd.read_csv(bboxes_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:25:31.926904Z","iopub.execute_input":"2026-01-08T14:25:31.927152Z","iopub.status.idle":"2026-01-08T14:25:32.080108Z","shell.execute_reply.started":"2026-01-08T14:25:31.927127Z","shell.execute_reply":"2026-01-08T14:25:32.079213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Columns trong bộ competition: \", s_csv.columns)\n# print(\"Columns trong bộ resized: \", p_df.columns)\nprint(\"Số lượng values trong bộ competition:\" , len(s_csv))\n# print(\"Số lượng values trong bộ resized:\" , len(p_df))\nprint(\"Số lượng ảnh trong bộ competition: \", s_csv['image_id'].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:25:32.081783Z","iopub.execute_input":"2026-01-08T14:25:32.082061Z","iopub.status.idle":"2026-01-08T14:25:32.104505Z","shell.execute_reply.started":"2026-01-08T14:25:32.082038Z","shell.execute_reply":"2026-01-08T14:25:32.103619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check class id của từng bệnh\nid_csv = s_csv[['class_name', 'class_id']].drop_duplicates().reset_index(drop=True)\nid_csv.sort_values(by='class_id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:25:32.105462Z","iopub.execute_input":"2026-01-08T14:25:32.105726Z","iopub.status.idle":"2026-01-08T14:25:32.143764Z","shell.execute_reply.started":"2026-01-08T14:25:32.105707Z","shell.execute_reply":"2026-01-08T14:25:32.143224Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ***2. Preprocessing***","metadata":{}},{"cell_type":"code","source":"train512_dir = '/kaggle/input/vinbigdata/train'\n# train256_dir = '/kaggle/input/vinbigdata-chest-xray-resized-png-256x256/train'\n\npath512_list = glob.glob(train512_dir + '/*')\n# path256_list = glob.glob(train256_dir + '/*')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:25:32.144447Z","iopub.execute_input":"2026-01-08T14:25:32.144672Z","iopub.status.idle":"2026-01-08T14:25:32.629084Z","shell.execute_reply.started":"2026-01-08T14:25:32.144655Z","shell.execute_reply":"2026-01-08T14:25:32.628521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add thêm path của từng ảnh vô trong file csv (512x512)\np512_csv = s_csv.copy()\np512_csv['image_path'] = p512_csv['image_id'].progress_apply(lambda x: next(filter(lambda y: x in y, path512_list), None))\np512_csv['image_path'].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:25:32.62978Z","iopub.execute_input":"2026-01-08T14:25:32.629986Z","iopub.status.idle":"2026-01-08T14:26:30.303988Z","shell.execute_reply.started":"2026-01-08T14:25:32.629962Z","shell.execute_reply":"2026-01-08T14:26:30.303444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(p512_csv)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:30.306177Z","iopub.execute_input":"2026-01-08T14:26:30.306388Z","iopub.status.idle":"2026-01-08T14:26:30.311031Z","shell.execute_reply.started":"2026-01-08T14:26:30.306372Z","shell.execute_reply":"2026-01-08T14:26:30.310319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# lung_diseases = { \"Atelectasis\" : 0,\n#                  \"Consolidation\" : 1, \n#                  \"ILD\" : 2, \n#                  \"Infiltration\" : 3, \n#                  \"Lung Opacity\" : 4, \n#                  \"Nodule/Mass\" : 5, \n#                  \"Pleural effusion\" : 6, \n#                  \"Pleural thickening\" : 7, \n#                  \"Pneumothorax\" : 8, \n#                  \"Pulmonary fibrosis\" : 9 \n#                 } \n\n# train512_df = p512_csv[p512_csv['class_name'].isin(lung_diseases)].reset_index(drop=True) \ntrain512_df = p512_csv[p512_csv['class_name']!='No finding'].reset_index(drop=True) \n# Tạo cột class_id mới theo dict\n# train512_df[\"class_id\"] = train512_df[\"class_name\"].map(lung_diseases)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:30.31185Z","iopub.execute_input":"2026-01-08T14:26:30.312127Z","iopub.status.idle":"2026-01-08T14:26:30.334575Z","shell.execute_reply.started":"2026-01-08T14:26:30.312108Z","shell.execute_reply":"2026-01-08T14:26:30.334042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train512_df['class_name'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:30.335227Z","iopub.execute_input":"2026-01-08T14:26:30.335422Z","iopub.status.idle":"2026-01-08T14:26:30.341754Z","shell.execute_reply.started":"2026-01-08T14:26:30.335406Z","shell.execute_reply":"2026-01-08T14:26:30.341113Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### *Ảnh gốc*","metadata":{}},{"cell_type":"code","source":"# Plot một vài ảnh để kiểm tra 512x512\nimgs512 = list(set(train512_df['image_path']))\ntest512_list = imgs512[:4]\n\nimg512_list = [cv2.imread(img) for img in test512_list]\nplot_img(img512_list)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:30.343182Z","iopub.execute_input":"2026-01-08T14:26:30.343433Z","iopub.status.idle":"2026-01-08T14:26:31.217924Z","shell.execute_reply.started":"2026-01-08T14:26:30.343408Z","shell.execute_reply":"2026-01-08T14:26:31.217141Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## *Lưu ảnh đã được xử lý*","metadata":{}},{"cell_type":"code","source":"# # Lấy list path của mỗi ảnh 512x512 và 256x256\nimg512_path_list = list(set(train512_df['image_path'].tolist()))\n# img256_path_list = list(set(train256_df['image_path'].tolist()))\nprint(\"512x512: \", len(img512_path_list))\n# print(\"256x256: \",len(img256_path_list))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:31.21883Z","iopub.execute_input":"2026-01-08T14:26:31.219056Z","iopub.status.idle":"2026-01-08T14:26:31.225129Z","shell.execute_reply.started":"2026-01-08T14:26:31.219036Z","shell.execute_reply":"2026-01-08T14:26:31.224412Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## *Lấy size gốc các ảnh được dùng trong training*","metadata":{}},{"cell_type":"code","source":"img_size_path = '/kaggle/input/vinbigdata/train_meta.csv'\nimg_size_df = pd.read_csv(img_size_path)\nimg_size_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:31.22596Z","iopub.execute_input":"2026-01-08T14:26:31.2262Z","iopub.status.idle":"2026-01-08T14:26:31.261733Z","shell.execute_reply.started":"2026-01-08T14:26:31.226175Z","shell.execute_reply":"2026-01-08T14:26:31.261209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train512_merge_df = train512_df.merge(img_size_df, on='image_id', how='left')\ntrain512_merge_df = train512_merge_df[['image_id', 'image_path', 'class_name', 'class_id', 'rad_id', 'x_min', 'y_min','x_max','y_max', 'dim0', 'dim1']]\ntrain512_merge_df = train512_merge_df.rename(columns = {\n    'dim0': 'height',\n    'dim1': 'width'\n})\ntrain512_merge_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:31.262366Z","iopub.execute_input":"2026-01-08T14:26:31.262596Z","iopub.status.idle":"2026-01-08T14:26:31.294433Z","shell.execute_reply.started":"2026-01-08T14:26:31.26258Z","shell.execute_reply":"2026-01-08T14:26:31.293935Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## *Resize lại bounding box cho đúng với ảnh resize*","metadata":{}},{"cell_type":"code","source":"# train1024_merge_df[['x_min_new', 'y_min_new', 'x_max_new', 'y_max_new']] = train1024_merge_df.progress_apply(resize_boxes, axis=1, result_type='expand', args=(1024, 1024))\ntrain512_merge_df[['x_min_new', 'y_min_new', 'x_max_new', 'y_max_new']] = train512_merge_df.progress_apply(resize_boxes, axis=1, result_type='expand', args=(512, 512))\n# train256_merge_df[['x_min_new', 'y_min_new', 'x_max_new', 'y_max_new']] = train256_merge_df.progress_apply(resize_boxes, axis=1, result_type='expand', args=(256, 256))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:31.295237Z","iopub.execute_input":"2026-01-08T14:26:31.295708Z","iopub.status.idle":"2026-01-08T14:26:35.717139Z","shell.execute_reply.started":"2026-01-08T14:26:31.29569Z","shell.execute_reply":"2026-01-08T14:26:35.716498Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## ***Visualize ảnh XRAY với bounding box đã được resize***","metadata":{}},{"cell_type":"code","source":"import random\nnum_images = 9\n\nlist_images = train512_merge_df['image_id'].tolist()\nsample_images = random.sample(list(list_images), num_images)\nsample_images","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:35.717797Z","iopub.execute_input":"2026-01-08T14:26:35.718068Z","iopub.status.idle":"2026-01-08T14:26:35.723708Z","shell.execute_reply.started":"2026-01-08T14:26:35.718045Z","shell.execute_reply":"2026-01-08T14:26:35.723026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dicom_dir = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train\"     # thư mục chứa DICOM\nsample_images_dicom = [i + \".dicom\" for i in sample_images]\n\nfull_paths = [os.path.join(dicom_dir, f) for f in sample_images_dicom]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:35.724345Z","iopub.execute_input":"2026-01-08T14:26:35.724549Z","iopub.status.idle":"2026-01-08T14:26:35.737408Z","shell.execute_reply.started":"2026-01-08T14:26:35.724525Z","shell.execute_reply":"2026-01-08T14:26:35.736749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list_dicom = [dicom2array(i) for i in full_paths]\nlen(list_dicom)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:35.738067Z","iopub.execute_input":"2026-01-08T14:26:35.738295Z","iopub.status.idle":"2026-01-08T14:26:46.255284Z","shell.execute_reply.started":"2026-01-08T14:26:35.738276Z","shell.execute_reply":"2026-01-08T14:26:46.253902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filenames = [os.path.basename(p).split('.')[0] for p in full_paths]\nprint(filenames)\nfiltered = s_csv[s_csv[\"image_id\"].isin(filenames)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:46.256379Z","iopub.execute_input":"2026-01-08T14:26:46.25674Z","iopub.status.idle":"2026-01-08T14:26:46.269595Z","shell.execute_reply.started":"2026-01-08T14:26:46.256719Z","shell.execute_reply":"2026-01-08T14:26:46.268695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import math\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport os\n\ndef show_multiple_dicoms_with_boxes(dicom_paths, df, cols=3):\n    rows = math.ceil(len(dicom_paths) / cols)\n\n    fig, axes = plt.subplots(rows, cols, figsize=(cols * 6, rows * 6))\n\n    # Nếu chỉ có 1 hàng -> giữ axes dạng list\n    if rows == 1:\n        axes = [axes]\n\n    # Flatten axes để dễ xử lý\n    axes = np.array(axes).reshape(-1)\n\n    for idx, dicom_path in enumerate(dicom_paths):\n        ax = axes[idx]\n        filename = os.path.basename(dicom_path).split('.')[0]\n\n        # Lọc bbox\n        boxes = df[df[\"image_id\"] == filename]\n        # Load ảnh\n        img = dicom2array(dicom_path)\n\n        ax.imshow(img, cmap=\"gray\")\n        ax.set_title(filename)\n        ax.axis(\"off\")\n\n        # Vẽ bbox\n        for _, row in boxes.iterrows():\n            x1, y1, x2, y2 = row[\"x_min\"], row[\"y_min\"], row[\"x_max\"], row[\"y_max\"]\n            class_name = row.get(\"class_name\", \"\")\n\n            rect = patches.Rectangle(\n                (x1, y1), \n                x2 - x1, \n                y2 - y1, \n                linewidth=2, \n                edgecolor='red', \n                facecolor='none'\n            )\n            ax.add_patch(rect)\n            ax.text(x1, y1 - 5, class_name, color=\"yellow\", fontsize=10, backgroundcolor=\"black\")\n\n    # Nếu số ảnh không chia hết cho cols → tắt các ô rỗng\n    for j in range(idx + 1, len(axes)):\n        axes[j].axis(\"off\")\n\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:46.270654Z","iopub.execute_input":"2026-01-08T14:26:46.271347Z","iopub.status.idle":"2026-01-08T14:26:46.283958Z","shell.execute_reply.started":"2026-01-08T14:26:46.271326Z","shell.execute_reply":"2026-01-08T14:26:46.283297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_multiple_dicoms_with_boxes(full_paths, train512_merge_df, cols=3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:46.285163Z","iopub.execute_input":"2026-01-08T14:26:46.285466Z","iopub.status.idle":"2026-01-08T14:26:58.893676Z","shell.execute_reply.started":"2026-01-08T14:26:46.285441Z","shell.execute_reply":"2026-01-08T14:26:58.892526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_folder = \"/kaggle/input/vinbigdata/train\"\n\nfig, axes = plt.subplots(3, 3, figsize=(15, 15))\naxes = axes.flatten()\n\nfor ax, img_id in zip(axes, sample_images):\n    img_path = os.path.join(image_folder, f'{img_id}.png')\n\n    if not os.path.exists(img_path):\n        print(\"Không tìm thấy ảnh trong thư mục\")\n        continue\n\n    image = Image.open(img_path).convert(\"RGB\")\n    ax.imshow(image)\n    ax.axis(\"off\")\n    ax.set_title(img_id)\n\n    org_width, org_height = image.size\n\n    bboxes = train512_merge_df[train512_merge_df['image_id'] == img_id]\n\n    for _, row in bboxes.iterrows():\n        if pd.notna(row['x_min']):\n            x_min, x_max, y_min, y_max = row['x_min_new'],row['x_max_new'],row['y_min_new'],row['y_max_new']\n            width, height = x_max - x_min, y_max - y_min\n\n            rect = patches.Rectangle(\n                (x_min, y_min), width, height,\n                linewidth=2, edgecolor=\"red\", facecolor=\"none\"\n            )\n\n            ax.add_patch(rect)\n            ax.text(x_min, y_min - 5, row[\"class_name\"], color=\"yellow\", fontsize=8,\n                    bbox=dict(facecolor=\"black\", alpha=0.5, pad=1))\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:26:58.894793Z","iopub.execute_input":"2026-01-08T14:26:58.895148Z","iopub.status.idle":"2026-01-08T14:27:00.616324Z","shell.execute_reply.started":"2026-01-08T14:26:58.895112Z","shell.execute_reply":"2026-01-08T14:27:00.615243Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### ***Merge overlapping bounding boxes trước khi crop***","metadata":{}},{"cell_type":"markdown","source":"**Tại sao cần merge bounding boxes?**\n\nMedical imaging datasets thường có nhiều radiologists annotate cùng 1 ảnh → overlapping boxes cho cùng 1 lesion.\n\n**Strategy:**\n1. Group boxes theo `image_id` và `class_name`\n2. Tính IoU (Intersection over Union) giữa mọi cặp boxes\n3. Nếu IoU > 0.3 → merge thành 1 box lớn hơn bao quanh cả 2\n4. Lặp lại cho đến khi không còn boxes nào merge được\n\n**Benefits:**\n- ✅ Giảm redundancy (nhiều boxes cho 1 lesion)\n- ✅ Cleaner training data\n- ✅ Better bbox quality\n- ✅ Faster training (ít boxes hơn)","metadata":{}},{"cell_type":"code","source":"# 🔧 MERGE OVERLAPPING BOUNDING BOXES\n# Merge các bbox cùng class có IoU > 0.3 để giảm redundancy\n\ndef calculate_iou(box1, box2):\n    \"\"\"\n    Tính IoU (Intersection over Union) giữa 2 bounding boxes.\n    \n    Args:\n        box1, box2: dict với keys ['x_min_new', 'y_min_new', 'x_max_new', 'y_max_new']\n    \n    Returns:\n        float: IoU score (0-1)\n    \"\"\"\n    x1_min, y1_min = box1['x_min_new'], box1['y_min_new']\n    x1_max, y1_max = box1['x_max_new'], box1['y_max_new']\n    \n    x2_min, y2_min = box2['x_min_new'], box2['y_min_new']\n    x2_max, y2_max = box2['x_max_new'], box2['y_max_new']\n    \n    # Tính intersection\n    inter_x_min = max(x1_min, x2_min)\n    inter_y_min = max(y1_min, y2_min)\n    inter_x_max = min(x1_max, x2_max)\n    inter_y_max = min(y1_max, y2_max)\n    \n    if inter_x_min >= inter_x_max or inter_y_min >= inter_y_max:\n        return 0.0\n    \n    inter_area = (inter_x_max - inter_x_min) * (inter_y_max - inter_y_min)\n    \n    # Tính union\n    box1_area = (x1_max - x1_min) * (y1_max - y1_min)\n    box2_area = (x2_max - x2_min) * (y2_max - y2_min)\n    union_area = box1_area + box2_area - inter_area\n    \n    if union_area == 0:\n        return 0.0\n    \n    iou = inter_area / union_area\n    return iou\n\n\ndef merge_boxes(box1, box2):\n    \"\"\"\n    Merge 2 bounding boxes thành 1 box bao quanh cả 2.\n    \n    Args:\n        box1, box2: dict với bbox coordinates\n    \n    Returns:\n        dict: Merged bounding box\n    \"\"\"\n    merged = box1.copy()\n    merged['x_min_new'] = min(box1['x_min_new'], box2['x_min_new'])\n    merged['y_min_new'] = min(box1['y_min_new'], box2['y_min_new'])\n    merged['x_max_new'] = max(box1['x_max_new'], box2['x_max_new'])\n    merged['y_max_new'] = max(box1['y_max_new'], box2['y_max_new'])\n    return merged\n\n\ndef merge_overlapping_boxes(df, iou_threshold=0.3):\n    \"\"\"\n    Merge các bounding boxes cùng class có IoU > threshold.\n    \n    Args:\n        df: DataFrame chứa bounding boxes với columns ['image_id', 'class_name', 'x_min_new', ...]\n        iou_threshold: IoU threshold để merge (default 0.3)\n    \n    Returns:\n        DataFrame: DataFrame sau khi merge\n    \"\"\"\n    if df.empty:\n        return df\n    \n    print(f\"📊 Before merging: {len(df)} bounding boxes\")\n    \n    # Columns cần giữ lại\n    required_cols = ['image_id', 'image_path', 'class_name', 'class_id', 'rad_id', \n                     'x_min', 'y_min', 'x_max', 'y_max', 'height', 'width',\n                     'x_min_new', 'y_min_new', 'x_max_new', 'y_max_new']\n    \n    # Filter columns tồn tại\n    keep_cols = [col for col in required_cols if col in df.columns]\n    \n    merged_records = []\n    processed_indices = set()\n    \n    # Group by image_id và class_name\n    for (img_id, class_name), group in tqdm(df.groupby(['image_id', 'class_name']), \n                                             desc=\"Merging overlapping boxes\"):\n        group_indices = group.index.tolist()\n        \n        # Skip nếu chỉ có 1 box\n        if len(group_indices) == 1:\n            merged_records.append(group.iloc[0].to_dict())\n            processed_indices.update(group_indices)\n            continue\n        \n        # Convert to list of dicts cho dễ xử lý\n        boxes = group.to_dict('records')\n        box_indices = group_indices.copy()\n        merged_flags = [False] * len(boxes)\n        \n        # Merge boxes có IoU > threshold\n        for i in range(len(boxes)):\n            if merged_flags[i]:\n                continue\n            \n            current_box = boxes[i]\n            boxes_to_merge = [i]\n            \n            # Tìm tất cả boxes overlap với current box\n            for j in range(i + 1, len(boxes)):\n                if merged_flags[j]:\n                    continue\n                \n                iou = calculate_iou(current_box, boxes[j])\n                \n                if iou > iou_threshold:\n                    boxes_to_merge.append(j)\n                    merged_flags[j] = True\n            \n            # Merge tất cả boxes found\n            if len(boxes_to_merge) > 1:\n                # Merge iteratively\n                merged_box = boxes[boxes_to_merge[0]]\n                for idx in boxes_to_merge[1:]:\n                    merged_box = merge_boxes(merged_box, boxes[idx])\n                merged_records.append(merged_box)\n                processed_indices.update([box_indices[idx] for idx in boxes_to_merge])\n            else:\n                # Không merge, giữ nguyên\n                merged_records.append(current_box)\n                processed_indices.add(box_indices[i])\n    \n    # Tạo DataFrame mới\n    merged_df = pd.DataFrame(merged_records)\n    \n    # Đảm bảo columns order\n    final_cols = [col for col in keep_cols if col in merged_df.columns]\n    merged_df = merged_df[final_cols]\n    \n    print(f\"✅ After merging: {len(merged_df)} bounding boxes\")\n    print(f\"📉 Reduced: {len(df) - len(merged_df)} boxes ({(len(df) - len(merged_df)) / len(df) * 100:.1f}%)\")\n    \n    return merged_df\n\n\n# 🚀 Apply merge to train512_merge_df\nprint(\"=\"*80)\nprint(\"🔧 MERGING OVERLAPPING BOUNDING BOXES\")\nprint(\"=\"*80)\nprint(f\"IoU Threshold: 0.3\")\nprint(f\"Strategy: Merge boxes with same class and IoU > 0.3\")\nprint()\n\ntrain512_merge_df_original = train512_merge_df.copy()\ntrain512_merge_df = merge_overlapping_boxes(train512_merge_df, iou_threshold=0.3)\n\nprint(\"\\n📊 Merge Statistics by Class:\")\nfor class_name in sorted(train512_merge_df['class_name'].unique()):\n    original_count = len(train512_merge_df_original[train512_merge_df_original['class_name'] == class_name])\n    merged_count = len(train512_merge_df[train512_merge_df['class_name'] == class_name])\n    reduction = original_count - merged_count\n    print(f\"  {class_name:25s}: {original_count:5d} → {merged_count:5d} (-{reduction:4d}, {reduction/original_count*100:5.1f}%)\")\n\nprint(\"\\n💡 Benefits of merging:\")\nprint(\"  ✓ Reduced redundant overlapping boxes\")\nprint(\"  ✓ Cleaner training data\")\nprint(\"  ✓ Better bbox quality\")\nprint(\"  ✓ Faster training (fewer boxes)\")\nprint(\"=\"*80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:00.617473Z","iopub.execute_input":"2026-01-08T14:27:00.618047Z","iopub.status.idle":"2026-01-08T14:27:09.314163Z","shell.execute_reply.started":"2026-01-08T14:27:00.617996Z","shell.execute_reply":"2026-01-08T14:27:09.313535Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### ***Visualize kết quả merge***","metadata":{}},{"cell_type":"code","source":"# 📊 Visualize Before/After Merge cho một vài sample images\n# Chọn images có nhiều boxes để thấy rõ effect\n\ndef visualize_merge_comparison(image_id, df_before, df_after, image_folder, figsize=(16, 8)):\n    \"\"\"\n    Hiển thị comparison giữa bboxes trước và sau merge.\n    \"\"\"\n    fig, axes = plt.subplots(1, 2, figsize=figsize)\n    \n    # Load image\n    img_path = os.path.join(image_folder, f'{image_id}.png')\n    if not os.path.exists(img_path):\n        print(f\"⚠️ Image not found: {img_path}\")\n        return\n    \n    image = Image.open(img_path).convert(\"RGB\")\n    \n    # === BEFORE MERGE ===\n    ax_before = axes[0]\n    ax_before.imshow(image)\n    ax_before.set_title(f\"Before Merge: {image_id[:15]}\", fontsize=14, fontweight='bold')\n    ax_before.axis(\"off\")\n    \n    boxes_before = df_before[df_before['image_id'] == image_id]\n    for _, row in boxes_before.iterrows():\n        if pd.notna(row['x_min_new']):\n            x_min, x_max = row['x_min_new'], row['x_max_new']\n            y_min, y_max = row['y_min_new'], row['y_max_new']\n            width, height = x_max - x_min, y_max - y_min\n            \n            rect = patches.Rectangle(\n                (x_min, y_min), width, height,\n                linewidth=2, edgecolor=\"red\", facecolor=\"none\", alpha=0.7\n            )\n            ax_before.add_patch(rect)\n            ax_before.text(x_min, y_min - 5, row[\"class_name\"], \n                          color=\"yellow\", fontsize=8, fontweight='bold',\n                          bbox=dict(facecolor=\"red\", alpha=0.7, pad=2))\n    \n    # === AFTER MERGE ===\n    ax_after = axes[1]\n    ax_after.imshow(image)\n    ax_after.set_title(f\"After Merge: {image_id[:15]}\", fontsize=14, fontweight='bold')\n    ax_after.axis(\"off\")\n    \n    boxes_after = df_after[df_after['image_id'] == image_id]\n    for _, row in boxes_after.iterrows():\n        if pd.notna(row['x_min_new']):\n            x_min, x_max = row['x_min_new'], row['x_max_new']\n            y_min, y_max = row['y_min_new'], row['y_max_new']\n            width, height = x_max - x_min, y_max - y_min\n            \n            rect = patches.Rectangle(\n                (x_min, y_min), width, height,\n                linewidth=2, edgecolor=\"lime\", facecolor=\"none\", alpha=0.7\n            )\n            ax_after.add_patch(rect)\n            ax_after.text(x_min, y_min - 5, row[\"class_name\"], \n                         color=\"yellow\", fontsize=8, fontweight='bold',\n                         bbox=dict(facecolor=\"green\", alpha=0.7, pad=2))\n    \n    plt.tight_layout()\n    plt.show()\n    \n    # Print statistics\n    print(f\"\\n{'='*60}\")\n    print(f\"Image ID: {image_id}\")\n    print(f\"Boxes before merge: {len(boxes_before)}\")\n    print(f\"Boxes after merge:  {len(boxes_after)}\")\n    print(f\"Reduction:          {len(boxes_before) - len(boxes_after)} boxes\")\n    print(f\"{'='*60}\\n\")\n\n\n# Find images với nhiều boxes để visualize\nbox_counts = train512_merge_df_original.groupby('image_id').size()\nimages_with_many_boxes = box_counts[box_counts >= 3].sort_values(ascending=False).head(5).index.tolist()\n\nprint(\"🔍 Visualizing merge results for images with multiple boxes...\")\nprint(f\"Selected {min(3, len(images_with_many_boxes))} images to visualize\\n\")\n\nfor img_id in images_with_many_boxes[:3]:  # Visualize top 3\n    visualize_merge_comparison(\n        image_id=img_id,\n        df_before=train512_merge_df_original,\n        df_after=train512_merge_df,\n        image_folder=image_folder\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:09.31494Z","iopub.execute_input":"2026-01-08T14:27:09.315233Z","iopub.status.idle":"2026-01-08T14:27:11.764544Z","shell.execute_reply.started":"2026-01-08T14:27:09.315208Z","shell.execute_reply":"2026-01-08T14:27:11.7638Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## *Lấy tỉ lệ tương đối của bounding box*","metadata":{}},{"cell_type":"code","source":"# train1024_yolo = normalize_bbox(train1024_merge_df)\ntrain512_yolo = normalize_bbox(train512_merge_df)\n# train256_yolo = normalize_bbox(train256_merge_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:11.765486Z","iopub.execute_input":"2026-01-08T14:27:11.765706Z","iopub.status.idle":"2026-01-08T14:27:11.778405Z","shell.execute_reply.started":"2026-01-08T14:27:11.765689Z","shell.execute_reply":"2026-01-08T14:27:11.777779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train512_yolo.to_csv('train512.csv', index=False, encoding='utf-8-sig')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:11.782113Z","iopub.execute_input":"2026-01-08T14:27:11.782301Z","iopub.status.idle":"2026-01-08T14:27:12.23936Z","shell.execute_reply.started":"2026-01-08T14:27:11.782287Z","shell.execute_reply":"2026-01-08T14:27:12.238785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train512_yolo.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:12.240053Z","iopub.execute_input":"2026-01-08T14:27:12.240306Z","iopub.status.idle":"2026-01-08T14:27:12.261154Z","shell.execute_reply.started":"2026-01-08T14:27:12.240283Z","shell.execute_reply":"2026-01-08T14:27:12.260592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train512_yolo.to_csv(\"/kaggle/working/bboxes.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:12.261788Z","iopub.execute_input":"2026-01-08T14:27:12.262036Z","iopub.status.idle":"2026-01-08T14:27:12.685145Z","shell.execute_reply.started":"2026-01-08T14:27:12.261999Z","shell.execute_reply":"2026-01-08T14:27:12.684529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def write_yolo_labels(df: pd.DataFrame, label_dir: str):\n    \"\"\"Write YOLO txt files from dataframe with normalized coordinates.\"\"\"\n    os.makedirs(label_dir, exist_ok=True)\n\n    df = df.copy()\n    # Ensure image_id is string\n    df['image_id'] = df['image_id'].astype(str)\n    \n    for image_id, group in tqdm(df.groupby('image_id'), desc='Writing YOLO labels'):\n        lines: list[str] = []\n        for _, row in group.iterrows():\n            # Assuming normalized columns exist\n            if pd.isna(row['x_center_norm']): continue\n            \n            lines.append(\n                f\"{int(row['class_id'])} {row['x_center_norm']:.6f} {row['y_center_norm']:.6f} {row['bbox_width_norm']:.6f} {row['bbox_height_norm']:.6f}\"\n            )\n\n        label_path = os.path.join(label_dir, f\"{image_id}.txt\")\n        with open(label_path, 'w', encoding='utf-8') as f:\n            f.write('\\n'.join(lines))\n\n# Define label directory\nimg_label_dir = \"/kaggle/working/chest_detection/labels\"\n\n# Generate labels\nprint(f\"Generating YOLO labels in {img_label_dir}...\")\nwrite_yolo_labels(train512_yolo, img_label_dir)\nprint(\"✅ Labels generated successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:12.685837Z","iopub.execute_input":"2026-01-08T14:27:12.686064Z","iopub.status.idle":"2026-01-08T14:27:14.647944Z","shell.execute_reply.started":"2026-01-08T14:27:12.686046Z","shell.execute_reply":"2026-01-08T14:27:14.647076Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## *Load file txt lên xem thử*","metadata":{}},{"cell_type":"code","source":"list_label = glob.glob(img_label_dir + \"/*\")\n\nwith open(list_label[0]) as f:\n    file = f.read()\n    print(file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:14.648688Z","iopub.execute_input":"2026-01-08T14:27:14.648861Z","iopub.status.idle":"2026-01-08T14:27:14.661296Z","shell.execute_reply.started":"2026-01-08T14:27:14.648847Z","shell.execute_reply":"2026-01-08T14:27:14.660563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(list_label)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:14.662054Z","iopub.execute_input":"2026-01-08T14:27:14.662695Z","iopub.status.idle":"2026-01-08T14:27:14.669775Z","shell.execute_reply.started":"2026-01-08T14:27:14.662672Z","shell.execute_reply":"2026-01-08T14:27:14.669115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# class_order = sorted(lung_diseases.items(), key=lambda item: item[1])\nunique_classes = train512_yolo[['class_name', 'class_id']].drop_duplicates()\nclass_order = dict(zip(unique_classes['class_name'], unique_classes['class_id']))\nclass_dict = {idx: name for name, idx in class_order.items()}\nclass_names_ordered = list(class_order.keys())\nclass_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:14.670537Z","iopub.execute_input":"2026-01-08T14:27:14.671449Z","iopub.status.idle":"2026-01-08T14:27:14.684595Z","shell.execute_reply.started":"2026-01-08T14:27:14.671426Z","shell.execute_reply":"2026-01-08T14:27:14.684022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_order","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:14.685255Z","iopub.execute_input":"2026-01-08T14:27:14.685717Z","iopub.status.idle":"2026-01-08T14:27:14.690034Z","shell.execute_reply.started":"2026-01-08T14:27:14.685701Z","shell.execute_reply":"2026-01-08T14:27:14.689343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(class_dict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:14.690766Z","iopub.execute_input":"2026-01-08T14:27:14.690948Z","iopub.status.idle":"2026-01-08T14:27:14.700149Z","shell.execute_reply.started":"2026-01-08T14:27:14.690933Z","shell.execute_reply":"2026-01-08T14:27:14.699515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"saved_path = \"/kaggle/working/chest_detection/p512x512_imgs\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:14.700731Z","iopub.execute_input":"2026-01-08T14:27:14.700911Z","iopub.status.idle":"2026-01-08T14:27:14.710482Z","shell.execute_reply.started":"2026-01-08T14:27:14.700897Z","shell.execute_reply":"2026-01-08T14:27:14.709811Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ***4. Model Implementation***","metadata":{}},{"cell_type":"markdown","source":"## *Chia dữ liệu thành tập train và validate*","metadata":{}},{"cell_type":"code","source":"# ⚠️ DEPRECATED: Simple split - này sẽ bị OVERRIDE bởi Stratified Split bên dưới\n# Keeping for reference only\n\nunique_img_ids = train512_merge_df['image_id'].unique()\nprint(f\"Total unique images: {len(unique_img_ids)}\")\nprint(\"⚠️ Note: Simple split below will be OVERRIDDEN by Stratified Split\")\nprint(\"=\" * 80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:14.711145Z","iopub.execute_input":"2026-01-08T14:27:14.711708Z","iopub.status.idle":"2026-01-08T14:27:14.722778Z","shell.execute_reply.started":"2026-01-08T14:27:14.711686Z","shell.execute_reply":"2026-01-08T14:27:14.722242Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### ***ADVANCED: Stratified Split để đảm bảo class balance***","metadata":{}},{"cell_type":"code","source":"# 🎯 STRATIFIED SPLIT: Đảm bảo mỗi class được phân bố đồng đều trong train/val\n# Điều này quan trọng cho imbalanced medical datasets\n\nfrom collections import Counter\n\n# Tạo label cho mỗi image (dominant class)\nimage_labels = {}\nunique_img_ids = train512_merge_df['image_id'].unique()\n\nfor image_id in unique_img_ids:\n    image_classes = train512_merge_df[train512_merge_df['image_id'] == image_id]['class_name'].tolist()\n    # Lấy class xuất hiện nhiều nhất\n    if image_classes:\n        dominant_class = Counter(image_classes).most_common(1)[0][0]\n        image_labels[image_id] = dominant_class\n\n# Convert to lists for stratification\nimage_ids = list(image_labels.keys())\nlabels = [image_labels[img_id] for img_id in image_ids]\n\n# Stratified split\nfrom sklearn.model_selection import StratifiedShuffleSplit\n\nsplitter = StratifiedShuffleSplit(n_splits=1, test_size=0.15, random_state=42)\ntrain_idx, val_idx = next(splitter.split(image_ids, labels))\n\ntrain_image_ids_stratified = [image_ids[i] for i in train_idx]\nval_image_ids_stratified = [image_ids[i] for i in val_idx]\n\nprint(\"📊 Stratified Split Results:\")\nprint(\"=\"*80)\nprint(f\"Train images: {len(train_image_ids_stratified)}\")\nprint(f\"Val images:   {len(val_image_ids_stratified)}\")\n\n# Verify class distribution\nprint(\"\\n📈 Class Distribution in Train Set:\")\ntrain_labels = [image_labels[img_id] for img_id in train_image_ids_stratified]\ntrain_dist = Counter(train_labels)\nfor class_name, count in sorted(train_dist.items()):\n    print(f\"  {class_name:25s}: {count:4d} ({count/len(train_labels)*100:.1f}%)\")\n\nprint(\"\\n📈 Class Distribution in Val Set:\")\nval_labels = [image_labels[img_id] for img_id in val_image_ids_stratified]\nval_dist = Counter(val_labels)\nfor class_name, count in sorted(val_dist.items()):\n    print(f\"  {class_name:25s}: {count:4d} ({count/len(val_labels)*100:.1f}%)\")\n\nprint(\"\\n✅ Stratified split ensures balanced class representation!\")\n\n# ✅ CRITICAL FIX: Override variables để các cells sau dùng stratified split\ntrain_image_ids = train_image_ids_stratified\nval_image_ids = val_image_ids_stratified\n\nprint(\"\\n🔧 FIXED: Overriding simple split with stratified split\")\nprint(f\"  ✅ train_image_ids now points to stratified split ({len(train_image_ids)} images)\")\nprint(f\"  ✅ val_image_ids now points to stratified split ({len(val_image_ids)} images)\")\nprint(\"=\"*80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:14.723382Z","iopub.execute_input":"2026-01-08T14:27:14.723552Z","iopub.status.idle":"2026-01-08T14:27:23.036803Z","shell.execute_reply.started":"2026-01-08T14:27:14.723539Z","shell.execute_reply":"2026-01-08T14:27:23.036058Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## *Phân tích Class Distribution để xử lý imbalance*","metadata":{}},{"cell_type":"code","source":"class_dict.items()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:23.037747Z","iopub.execute_input":"2026-01-08T14:27:23.037996Z","iopub.status.idle":"2026-01-08T14:27:23.042791Z","shell.execute_reply.started":"2026-01-08T14:27:23.037972Z","shell.execute_reply":"2026-01-08T14:27:23.042065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Phân tích class distribution\nclass_distribution = train512_merge_df['class_name'].value_counts()\nprint(\"Class Distribution:\")\nprint(class_distribution)\nprint(\"\\n\" + \"=\"*60)\n\n# Tính class weights để xử lý imbalance (inverse frequency)\ntotal_samples = len(train512_merge_df)\nclass_weights = {}\nfor class_name, count in class_distribution.items():\n    class_id = class_order[class_name]\n    weight = total_samples / (len(class_dict) * count)\n    class_weights[class_id] = round(weight, 3)\n\nprint(\"\\nClass Weights (for loss balancing):\")\nfor class_name, class_id in sorted(class_order.items(), key=lambda x: x[1]):\n    print(f\"{class_name:25s} (ID {class_id}): {class_weights[class_id]:.3f}\")\n\n# Visualize distribution\nplt.figure(figsize=(12, 6))\nclass_distribution.plot(kind='bar', color='steelblue')\nplt.title('Class Distribution in Dataset', fontsize=14, fontweight='bold')\nplt.xlabel('Disease Class')\nplt.ylabel('Number of Instances')\nplt.xticks(rotation=45, ha='right')\nplt.grid(axis='y', alpha=0.3)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:23.043539Z","iopub.execute_input":"2026-01-08T14:27:23.043742Z","iopub.status.idle":"2026-01-08T14:27:23.320623Z","shell.execute_reply.started":"2026-01-08T14:27:23.043724Z","shell.execute_reply":"2026-01-08T14:27:23.319902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tạo folder chia dữ liệu thành images và labels \nos.makedirs(\"/kaggle/working/dataset/train/images\", exist_ok=True)\nos.makedirs(\"/kaggle/working/dataset/val/images\", exist_ok=True)\nos.makedirs(\"/kaggle/working/dataset/train/labels\", exist_ok=True)\nos.makedirs(\"/kaggle/working/dataset/val/labels\", exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:23.32136Z","iopub.execute_input":"2026-01-08T14:27:23.321606Z","iopub.status.idle":"2026-01-08T14:27:23.32642Z","shell.execute_reply.started":"2026-01-08T14:27:23.32159Z","shell.execute_reply":"2026-01-08T14:27:23.325552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🎨 APPLY CLAHE PREPROCESSING\n# Apply Contrast Limited Adaptive Histogram Equalization to all images\nfrom skimage import exposure\n\n# Define CLAHE output folder\nclahe_output_folder = \"/kaggle/working/clahe_images\"\nos.makedirs(clahe_output_folder, exist_ok=True)\n\n# Get unique image paths from the dataframe\n# We use the 'image_path' column which should point to the source images\nunique_image_paths = train512_merge_df['image_path'].unique().tolist()\n\nprint(f\"Applying CLAHE to {len(unique_image_paths)} images...\")\nprint(f\"Source images example: {unique_image_paths[0]}\")\nprint(f\"Output folder: {clahe_output_folder}\")\n\n# Run CLAHE in parallel\nsave_img(unique_image_paths, clahe_output_folder)\n\nprint(f\"✅ CLAHE processing complete. Images saved to {clahe_output_folder}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:27:23.327201Z","iopub.execute_input":"2026-01-08T14:27:23.327422Z","iopub.status.idle":"2026-01-08T14:28:04.722968Z","shell.execute_reply.started":"2026-01-08T14:27:23.327399Z","shell.execute_reply":"2026-01-08T14:28:04.722352Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### ***Create DataFrames with CLAHE images và stratified split***","metadata":{}},{"cell_type":"code","source":"# ✅ FIXED: Create DataFrames với CLAHE images và stratified split\n# train_image_ids và val_image_ids đã được override bởi stratified version ở cell trước\n\n# Verify variables are correct\nprint(\"🔍 Verifying variables:\")\nprint(f\"  train_image_ids length: {len(train_image_ids)}\")\nprint(f\"  val_image_ids length: {len(val_image_ids)}\")\nprint(f\"  CLAHE folder exists: {os.path.exists(clahe_output_folder)}\")\n# Check count of generated images\nclahe_files = glob.glob(os.path.join(clahe_output_folder, '*.png'))\nprint(f\"  CLAHE images count: {len(clahe_files)}\")\nprint()\n\n# Create DataFrames using CLAHE images\n# The images are now in clahe_output_folder with name {image_id}.png\ntrain_labels = [os.path.join(img_label_dir, f\"{img_id}.txt\") for img_id in train_image_ids]\ntrain_images = [os.path.join(clahe_output_folder, f\"{img_id}.png\") for img_id in train_image_ids]\nyolo_df_train = pd.DataFrame({'label_path': train_labels, 'image_path': train_images})\n\nval_labels = [os.path.join(img_label_dir, f\"{img_id}.txt\") for img_id in val_image_ids]\nval_images = [os.path.join(clahe_output_folder, f\"{img_id}.png\") for img_id in val_image_ids]\nyolo_df_val = pd.DataFrame({'label_path': val_labels, 'image_path': val_images})\n\nprint(\"✅ DataFrames created successfully!\")\nprint(f\"  📁 Image source: {clahe_output_folder} (CLAHE enhanced)\")\nprint(f\"  📊 Split method: Stratified (balanced classes)\")\nprint(f\"  📈 Train samples: {len(yolo_df_train)}\")\nprint(f\"  📉 Val samples:   {len(yolo_df_val)}\")\n\n# Verify no data leakage\ntrain_ids_set = set(train_image_ids)\nval_ids_set = set(val_image_ids)\noverlap = train_ids_set.intersection(val_ids_set)\nprint(f\"\\n🔒 Data leakage check: {len(overlap)} overlapping images\")\nif len(overlap) == 0:\n    print(\"  ✅ PASSED - No data leakage!\")\nelse:\n    print(f\"  ❌ FAILED - Found {len(overlap)} overlapping images!\")\n\n# Verify files exist\ntrain_missing = sum(1 for p in train_images if not os.path.exists(p))\nval_missing = sum(1 for p in val_images if not os.path.exists(p))\nif train_missing + val_missing == 0:\n    print(f\"\\n✅ All image files exist!\")\nelse:\n    print(f\"\\n⚠️ Warning: {train_missing} train + {val_missing} val images missing\")\n    # Print first missing file for debugging\n    for p in train_images:\n        if not os.path.exists(p):\n            print(f\"  Missing train image: {p}\")\n            break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:28:04.723726Z","iopub.execute_input":"2026-01-08T14:28:04.723988Z","iopub.status.idle":"2026-01-08T14:28:04.768126Z","shell.execute_reply.started":"2026-01-08T14:28:04.72397Z","shell.execute_reply":"2026-01-08T14:28:04.767575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yolo_df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:28:04.768826Z","iopub.execute_input":"2026-01-08T14:28:04.769088Z","iopub.status.idle":"2026-01-08T14:28:04.775559Z","shell.execute_reply.started":"2026-01-08T14:28:04.769069Z","shell.execute_reply":"2026-01-08T14:28:04.775021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Copying train images\nprint(\"COPYING TRAIN IMAGES :-->\", \"-\"*50)\nfor img_path in yolo_df_train['image_path']:\n    shutil.copy(img_path, \"/kaggle/working/dataset/train/images\")\n    \n# Copying validation images\nprint(\"COPYING VALID IMAGES :-->\", \"-\"*50)\nfor img_path in yolo_df_val['image_path']:\n    shutil.copy(img_path, \"/kaggle/working/dataset/val/images\")\n\n# Copying train labels\nprint(\"COPYING TRAIN LABELS :-->\", \"-\"*50)\nfor label_path in yolo_df_train['label_path']:\n    shutil.copy(label_path, \"/kaggle/working/dataset/train/labels\")\n\n# Copying validation labels\nprint(\"COPYING VALID LABELS :-->\", \"-\"*50)\nfor label_path in yolo_df_val['label_path']:\n    shutil.copy(label_path, \"/kaggle/working/dataset/val/labels\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:28:04.776238Z","iopub.execute_input":"2026-01-08T14:28:04.776471Z","iopub.status.idle":"2026-01-08T14:28:05.85892Z","shell.execute_reply.started":"2026-01-08T14:28:04.776455Z","shell.execute_reply":"2026-01-08T14:28:05.85834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_label_count = len(glob.glob('/kaggle/working/dataset/train/labels/*'))\nprint(\"Number of train labels:\", train_label_count)\n\ntrain_image_count = len(glob.glob('/kaggle/working/dataset/train/images/*'))\nprint(\"Number of train images:\", train_image_count)\n\nvalid_label_count = len(glob.glob('/kaggle/working/dataset/val/labels/*'))\nprint(\"Number of valid labels:\", valid_label_count)\n\nvalid_image_count = len(glob.glob('/kaggle/working/dataset/val/images/*'))\nprint(\"Number of valid images:\", valid_image_count)\n\n# Verify counts match\nassert train_label_count == train_image_count == len(yolo_df_train), \"Train label/image count mismatch!\"\nassert valid_label_count == valid_image_count == len(yolo_df_val), \"Val label/image count mismatch!\"\nprint(\"\\n✅ All counts verified successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:28:05.85965Z","iopub.execute_input":"2026-01-08T14:28:05.859882Z","iopub.status.idle":"2026-01-08T14:28:05.880668Z","shell.execute_reply.started":"2026-01-08T14:28:05.859857Z","shell.execute_reply":"2026-01-08T14:28:05.879957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"move_dir = '/kaggle/working/dataset'\ngoal_dir = '/kaggle/working/yolo/dataset'\n\nshutil.move(move_dir, goal_dir)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:28:05.881357Z","iopub.execute_input":"2026-01-08T14:28:05.881522Z","iopub.status.idle":"2026-01-08T14:28:07.236274Z","shell.execute_reply.started":"2026-01-08T14:28:05.88151Z","shell.execute_reply":"2026-01-08T14:28:07.235657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yaml_dir = \"/kaggle/working/yolo/dataset\"\n\ndata = {\n    'names': class_names_ordered,\n    'nc': len(class_names_ordered),\n\n    'train': '/kaggle/working/yolo/dataset/train/images/',\n    'val': '/kaggle/working/yolo/dataset/val/images/'\n}\n\nwith open(yaml_dir+'/data.yaml', 'w') as file:\n    yaml.dump(data, file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:28:07.236979Z","iopub.execute_input":"2026-01-08T14:28:07.23737Z","iopub.status.idle":"2026-01-08T14:28:07.243413Z","shell.execute_reply.started":"2026-01-08T14:28:07.237348Z","shell.execute_reply":"2026-01-08T14:28:07.242768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from ultralytics import YOLO\n\n# # Load model (yolo12s.pt)\n# model = YOLO('yolo12s.pt')\n\n# # Info\n# print(\"=\"*80)\n# print(\"MODEL ARCHITECTURE INFO:\")\n# print(\"=\"*80)\n# model.info()\n# print(\"\\n\")\n\n# # ---------------------------\n# # TRAIN WITH DEFAULT SETTINGS\n# # ---------------------------\n# print(\"=\"*80)\n# print(\"STARTING YOLO DEFAULT TRAINING...\")\n# print(\"=\"*80)\n\n# results = model.train(\n#     data=yaml_dir + '/data.yaml',\n\n#     # Common training config\n#     imgsz=512,\n#     epochs=150,\n#     batch=16,\n#     device=0,\n\n#     # Only customization you requested\n#     patience=10,          # EARLY STOPPING = 10\n\n#     # Everything else = DEFAULT!!!\n#     project='/kaggle/working/runs',\n#     name='yolo12s_default',\n#     exist_ok=True\n# )\n\n# print(\"\\n\" + \"=\"*80)\n# print(\"TRAINING COMPLETED!\")\n# print(\"=\"*80)\n\n# # --------------------\n# # VALIDATION (DEFAULT)\n# # --------------------\n# print(\"\\n\" + \"=\"*80)\n# print(\"FINAL VALIDATION RESULTS:\")\n# print(\"=\"*80)\n\n# metrics = model.val()\n\n# print(f\"\\n📊 Overall Metrics:\")\n# print(f\"  mAP50-95:  {metrics.box.map:.4f}\")\n# print(f\"  mAP50:     {metrics.box.map50:.4f}\")\n# print(f\"  mAP75:     {metrics.box.map75:.4f}\")\n# print(f\"  Precision: {metrics.box.mp:.4f}\")\n# print(f\"  Recall:    {metrics.box.mr:.4f}\")\n\n# # Save results\n# results_summary = {\n#     'mAP50_95': float(metrics.box.map),\n#     'mAP50': float(metrics.box.map50),\n#     'mAP75': float(metrics.box.map75),\n#     'precision': float(metrics.box.mp),\n#     'recall': float(metrics.box.mr),\n# }\n\n# import json\n# with open('/kaggle/working/final_metrics.json', 'w') as f:\n#     json.dump(results_summary, f, indent=2)\n\n# print(\"\\n✅ Metrics saved to: /kaggle/working/final_metrics.json\")\n# print(\"✅ Best model saved to:\", model.trainer.best)\n# print(\"=\"*80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:28:07.244102Z","iopub.execute_input":"2026-01-08T14:28:07.244273Z","iopub.status.idle":"2026-01-08T14:28:07.255293Z","shell.execute_reply.started":"2026-01-08T14:28:07.244259Z","shell.execute_reply":"2026-01-08T14:28:07.254576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from ultralytics import YOLO\n\n# # 1. Load mô hình YOLO với weight đã train\n# model = YOLO(\"/kaggle/working/runs/yolo12s_default/weights/best.pt\")\n\n# # 2. Evaluate mô hình trên test set\n# results = model.val(\n#     data=yaml_dir+'/data.yaml',\n#     split=\"val\",\n#     imgsz=512,\n#     conf=0.25,\n#     iou=0.5\n# )\n\n# # 3. Lấy metric\n# map50 = results.box.map50\n# map5095 = results.box.map\n# precision = results.box.mean_results()[0]  # Precision\n# recall = results.box.mean_results()[1]     # Recall\n\n# # 4. In kết quả\n# print(f\"mAP50: {map50:.4f}\")\n# print(f\"mAP50-95: {map5095:.4f}\")\n# print(f\"Precision: {precision:.4f}\")\n# print(f\"Recall: {recall:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:28:07.256048Z","iopub.execute_input":"2026-01-08T14:28:07.256281Z","iopub.status.idle":"2026-01-08T14:28:07.267803Z","shell.execute_reply.started":"2026-01-08T14:28:07.256256Z","shell.execute_reply":"2026-01-08T14:28:07.26708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from ultralytics import YOLO\n\n# 1. Load mô hình YOLO với weight đã train\nmodel = YOLO(\"/kaggle/input/yolo-12-small-weight/Yolov12s_CLAHE_WBF_150ep.pt\")\n\n# 2. Evaluate mô hình trên test set\nresults = model.val(\n    data=yaml_dir+'/data.yaml',\n    split=\"val\",\n    imgsz=512,\n    conf=0.15,\n    iou=0.5\n)\n\n# 3. Lấy metric\nmap50 = results.box.map50\nmap5095 = results.box.map\nprecision = results.box.mean_results()[0]  # Precision\nrecall = results.box.mean_results()[1]     # Recall\n\n# 4. In kết quả\nprint(f\"mAP50: {map50:.4f}\")\nprint(f\"mAP50-95: {map5095:.4f}\")\nprint(f\"Precision: {precision:.4f}\")\nprint(f\"Recall: {recall:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T14:34:39.456549Z","iopub.execute_input":"2026-01-08T14:34:39.457163Z","iopub.status.idle":"2026-01-08T14:34:52.624344Z","shell.execute_reply.started":"2026-01-08T14:34:39.457133Z","shell.execute_reply":"2026-01-08T14:34:52.623704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}