{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================================================\n# VINBIGDATA -> YOLO FAST PIPELINE (OPTIMIZED)\n# =========================================================\n\nimport os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport random\nfrom tqdm import tqdm\nfrom multiprocessing import Pool, cpu_count\n\n# =========================================================\n# CONFIG\n# =========================================================\nIMG_SIZE = 1024\nIOU_THRESHOLD = 0.5\nNEG_RATIO = 0.30  # Giữ lại 30% ảnh \"No Finding\" để tránh mất cân bằng dữ liệu\n\nINPUT_DIR = \"/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection\"\nOUTPUT_DIR = \"/kaggle/working/vinbig_yolo_fast\"\n\nTRAIN_DIR = os.path.join(INPUT_DIR, \"train\")\nTRAIN_IMG_OUT = os.path.join(OUTPUT_DIR, \"images/train\")\nTRAIN_LAB_OUT = os.path.join(OUTPUT_DIR, \"labels/train\")\n\nos.makedirs(TRAIN_IMG_OUT, exist_ok=True)\nos.makedirs(TRAIN_LAB_OUT, exist_ok=True)\n\n# =========================================================\n# LOAD & SAMPLE DATA\n# =========================================================\nCSV_PATH = os.path.join(INPUT_DIR, \"train.csv\")\ndf = pd.read_csv(CSV_PATH)\n\n# Tách riêng ảnh bình thường (14) và ảnh bệnh lý\ndf_pos = df[df[\"class_id\"] != 14].copy()\ndf_neg = df[df[\"class_id\"] == 14].copy()\n\nneg_imgs = df_neg[\"image_id\"].unique()\nrandom.seed(42)\n\n# Lấy mẫu ngẫu nhiên Class 14\nkeep_neg_imgs = set(random.sample(list(neg_imgs), int(len(neg_imgs) * NEG_RATIO)))\ndf_neg = df_neg[df_neg[\"image_id\"].isin(keep_neg_imgs)]\n\ndf = pd.concat([df_pos, df_neg], axis=0)\ngrouped_df = {img_id: g for img_id, g in df.groupby(\"image_id\")}\nimage_ids = list(grouped_df.keys())\n\nprint(f\"Total images to process: {len(image_ids)}\")\n\n# =========================================================\n# HELPERS\n# =========================================================\ndef iou(b1, b2):\n    x1 = max(b1[0], b2[0])\n    y1 = max(b1[1], b2[1])\n    x2 = min(b1[2], b2[2])\n    y2 = min(b1[3], b2[3])\n    inter = max(0, x2 - x1) * max(0, y2 - y1)\n    a1 = (b1[2]-b1[0]) * (b1[3]-b1[1])\n    a2 = (b2[2]-b2[0]) * (b2[3]-b2[1])\n    return inter / (a1 + a2 - inter + 1e-6)\n\ndef clip_coords(xc, yc, w, h):\n    \"\"\"Đảm bảo tọa độ luôn nằm trong [0, 1]\"\"\"\n    return max(0, min(1, xc)), max(0, min(1, yc)), max(0, min(1, w)), max(0, min(1, h))\n\ndef merge_boxes(df_subset, thr=0.5):\n    results = []\n    for cls in df_subset[\"class_id\"].unique():\n        df_c = df_subset[df_subset[\"class_id\"] == cls]\n        boxes = df_c[[\"x_min\",\"y_min\",\"x_max\",\"y_max\"]].values.tolist()\n        used = [False]*len(boxes)\n        for i in range(len(boxes)):\n            if used[i]: continue\n            group = [boxes[i]]\n            used[i] = True\n            for j in range(i+1, len(boxes)):\n                if not used[j] and iou(boxes[i], boxes[j]) > thr:\n                    group.append(boxes[j])\n                    used[j] = True\n            group = np.array(group)\n            results.append([int(cls), group[:,0].mean(), group[:,1].mean(), group[:,2].mean(), group[:,3].mean()])\n    return results\n\ndef read_dicom(path):\n    ds = pydicom.dcmread(path)\n    img = ds.pixel_array.astype(np.float32)\n    # Chuẩn hóa 0-255\n    img = (img - img.min()) / (img.max() - img.min() + 1e-6)\n    img = (img * 255).astype(np.uint8)\n    # Xử lý ảnh âm bản (nếu có)\n    if ds.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 255 - img\n    return img\n\ndef letterbox(img, size=1024):\n    h, w = img.shape\n    scale = min(size/w, size/h)\n    nw, nh = int(w*scale), int(h*scale)\n    img = cv2.resize(img, (nw, nh))\n    canvas = np.full((size,size), 114, dtype=np.uint8)\n    top = (size-nh)//2\n    left = (size-nw)//2\n    canvas[top:top+nh, left:left+nw] = img\n    return canvas, scale, left, top\n\n# =========================================================\n# CORE PROCESS\n# =========================================================\ndef process(img_id):\n    try:\n        path = os.path.join(TRAIN_DIR, img_id + \".dicom\")\n        img = read_dicom(path)\n        \n        # 1. Resize & Letterbox\n        img, scale, left, top = letterbox(img, IMG_SIZE)\n        \n        # 2. Lưu ảnh JPG chất lượng cao (giảm tải dung lượng disk)\n        cv2.imwrite(\n            os.path.join(TRAIN_IMG_OUT, img_id + \".jpg\"), \n            img, \n            [int(cv2.IMWRITE_JPEG_QUALITY), 95]\n        )\n        \n        # 3. Xử lý nhãn\n        rows = grouped_df[img_id]\n        merged = merge_boxes(rows, IOU_THRESHOLD)\n        \n        lines = []\n        for c, x1, y1, x2, y2 in merged:\n            # Nếu là Class 14 (No Finding) -> File nhãn sẽ để trống theo chuẩn YOLO\n            if int(c) == 14:\n                continue\n            \n            # Scale tọa độ về ảnh mới\n            nx1 = x1 * scale + left\n            ny1 = y1 * scale + top\n            nx2 = x2 * scale + left\n            ny2 = y2 * scale + top\n            \n            # Chuyển về format YOLO (center_x, center_y, width, height)\n            xc = ((nx1 + nx2) / 2) / IMG_SIZE\n            yc = ((ny1 + ny2) / 2) / IMG_SIZE\n            w  = (nx2 - nx1) / IMG_SIZE\n            h  = (ny2 - ny1) / IMG_SIZE\n            \n            # Clip tọa độ để tránh lỗi > 1.0\n            xc, yc, w, h = clip_coords(xc, yc, w, h)\n            lines.append(f\"{int(c)} {xc:.6f} {yc:.6f} {w:.6f} {h:.6f}\")\n            \n        # 4. Lưu file label (.txt)\n        with open(os.path.join(TRAIN_LAB_OUT, img_id + \".txt\"), \"w\") as f:\n            f.write(\"\\n\".join(lines))\n            \n        return 1\n    except Exception as e:\n        # Không in print quá nhiều trong Pool tránh lag console\n        return 0\n\n# =========================================================\n# RUN MULTIPROCESSING\n# =========================================================\nif __name__ == \"__main__\":\n    n_cpu = max(cpu_count() - 1, 1)\n    print(f\"Using {n_cpu} CPUs\")\n    \n    with Pool(n_cpu) as p:\n        results = list(tqdm(p.imap(process, image_ids), total=len(image_ids)))\n    \n    print(f\"\\n✅ Xử lý hoàn tất: {sum(results)}/{len(image_ids)} ảnh.\")\n    print(f\"📁 Dữ liệu được lưu tại: {OUTPUT_DIR}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # =========================================================\n# # VINBIGDATA -> YOLO FAST PIPELINE\n# # WITH CONTROLLED CLASS 14 DOWNSAMPLING\n# # =========================================================\n\n# import os\n# import cv2\n# import numpy as np\n# import pandas as pd\n# import pydicom\n# import random\n\n# from tqdm import tqdm\n# from multiprocessing import Pool, cpu_count\n\n# # =========================================================\n# # CONFIG\n# # =========================================================\n# IMG_SIZE = 1024\n# IOU_THRESHOLD = 0.5\n\n# INPUT_DIR = \"/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection\"\n# OUTPUT_DIR = \"/kaggle/working/vinbig_yolo_fast\"\n\n# TRAIN_DIR = os.path.join(INPUT_DIR, \"train\")\n\n# TRAIN_IMG_OUT = os.path.join(OUTPUT_DIR, \"images/train\")\n# TRAIN_LAB_OUT = os.path.join(OUTPUT_DIR, \"labels/train\")\n\n# os.makedirs(TRAIN_IMG_OUT, exist_ok=True)\n# os.makedirs(TRAIN_LAB_OUT, exist_ok=True)\n\n# # =========================================================\n# # LOAD DATA\n# # =========================================================\n# CSV_PATH = os.path.join(INPUT_DIR, \"train.csv\")\n# df = pd.read_csv(CSV_PATH)\n\n# # =========================================================\n# # 🔥 CONTROL CLASS 14 RATIO\n# # =========================================================\n# NEG_RATIO = 0.30   # 👈 KHUYẾN NGHỊ: 30%\n\n# df_pos = df[df[\"class_id\"] != 14].copy()\n# df_neg = df[df[\"class_id\"] == 14].copy()\n\n# neg_imgs = df_neg[\"image_id\"].unique()\n\n# random.seed(42)\n\n# keep_neg_imgs = set(\n#     random.sample(\n#         list(neg_imgs),\n#         int(len(neg_imgs) * NEG_RATIO)\n#     )\n# )\n\n# df_neg = df_neg[df_neg[\"image_id\"].isin(keep_neg_imgs)]\n\n# df = pd.concat([df_pos, df_neg], axis=0)\n\n# print(\"Total annotations after sampling:\", len(df))\n\n# # =========================================================\n# # GROUP BY IMAGE\n# # =========================================================\n# grouped_df = {\n#     img_id: g\n#     for img_id, g in df.groupby(\"image_id\")\n# }\n\n# image_ids = list(grouped_df.keys())\n\n# print(\"Total images:\", len(image_ids))\n\n# # =========================================================\n# # IOU\n# # =========================================================\n# def iou(b1, b2):\n\n#     x1 = max(b1[0], b2[0])\n#     y1 = max(b1[1], b2[1])\n\n#     x2 = min(b1[2], b2[2])\n#     y2 = min(b1[3], b2[3])\n\n#     inter = max(0, x2 - x1) * max(0, y2 - y1)\n\n#     a1 = (b1[2]-b1[0]) * (b1[3]-b1[1])\n#     a2 = (b2[2]-b2[0]) * (b2[3]-b2[1])\n\n#     return inter / (a1 + a2 - inter + 1e-6)\n\n# # =========================================================\n# # MERGE BOXES\n# # =========================================================\n# def merge_boxes(df_subset, thr=0.5):\n\n#     results = []\n\n#     for cls in df_subset[\"class_id\"].unique():\n\n#         df_c = df_subset[df_subset[\"class_id\"] == cls]\n\n#         boxes = df_c[[\n#             \"x_min\",\"y_min\",\"x_max\",\"y_max\"\n#         ]].values.tolist()\n\n#         used = [False]*len(boxes)\n\n#         for i in range(len(boxes)):\n\n#             if used[i]:\n#                 continue\n\n#             group = [boxes[i]]\n#             used[i] = True\n\n#             for j in range(i+1, len(boxes)):\n\n#                 if not used[j] and iou(boxes[i], boxes[j]) > thr:\n\n#                     group.append(boxes[j])\n#                     used[j] = True\n\n#             group = np.array(group)\n\n#             results.append([\n#                 int(cls),\n#                 group[:,0].mean(),\n#                 group[:,1].mean(),\n#                 group[:,2].mean(),\n#                 group[:,3].mean()\n#             ])\n\n#     return results\n\n# # =========================================================\n# # DICOM READ\n# # =========================================================\n# def read_dicom(path):\n\n#     ds = pydicom.dcmread(path)\n\n#     img = ds.pixel_array.astype(np.float32)\n\n#     img = (img - img.min()) / (img.max() - img.min() + 1e-6)\n#     img = (img * 255).astype(np.uint8)\n\n#     if ds.PhotometricInterpretation == \"MONOCHROME1\":\n#         img = 255 - img\n\n#     return img\n\n# # =========================================================\n# # LETTERBOX\n# # =========================================================\n# def letterbox(img, size=640):\n\n#     h, w = img.shape\n\n#     scale = min(size/w, size/h)\n\n#     nw, nh = int(w*scale), int(h*scale)\n\n#     img = cv2.resize(img, (nw, nh))\n\n#     canvas = np.full((size,size), 114, dtype=np.uint8)\n\n#     top = (size-nh)//2\n#     left = (size-nw)//2\n\n#     canvas[top:top+nh, left:left+nw] = img\n\n#     return canvas, scale, left, top\n\n# # =========================================================\n# # PROCESS\n# # =========================================================\n# def process(img_id):\n\n#     try:\n\n#         path = os.path.join(TRAIN_DIR, img_id+\".dicom\")\n\n#         img = read_dicom(path)\n\n#         img, scale, left, top = letterbox(img, IMG_SIZE)\n\n#         cv2.imwrite(\n#             os.path.join(TRAIN_IMG_OUT, img_id+\".png\"),\n#             img,\n#             [cv2.IMWRITE_PNG_COMPRESSION, 3]\n#         )\n\n#         rows = grouped_df[img_id]\n\n#         merged = merge_boxes(rows, IOU_THRESHOLD)\n\n#         lines = []\n\n#         for c,x1,y1,x2,y2 in merged:\n\n#             x1 = x1*scale + left\n#             y1 = y1*scale + top\n#             x2 = x2*scale + left\n#             y2 = y2*scale + top\n\n#             xc = ((x1+x2)/2)/IMG_SIZE\n#             yc = ((y1+y2)/2)/IMG_SIZE\n#             w  = (x2-x1)/IMG_SIZE\n#             h  = (y2-y1)/IMG_SIZE\n\n#             lines.append(\n#                 f\"{c} {xc:.6f} {yc:.6f} {w:.6f} {h:.6f}\"\n#             )\n\n#         with open(\n#             os.path.join(TRAIN_LAB_OUT, img_id+\".txt\"),\n#             \"w\"\n#         ) as f:\n#             f.write(\"\\n\".join(lines))\n\n#         return 1\n\n#     except:\n#         return 0\n\n# # =========================================================\n# # MULTIPROCESS\n# # =========================================================\n# if __name__ == \"__main__\":\n\n#     n_cpu = max(cpu_count()-2, 1)\n\n#     print(\"CPUs:\", n_cpu)\n\n#     with Pool(n_cpu) as p:\n\n#         r = list(\n#             tqdm(\n#                 p.imap(process, image_ids),\n#                 total=len(image_ids)\n#             )\n#         )\n\n#     print(\"\\nDONE:\", sum(r), \"/\", len(r))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T08:14:03.979414Z","iopub.execute_input":"2026-05-12T08:14:03.979831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport random\n\n# =========================================================\n# CONFIG\n# =========================================================\nINPUT_DIR = \"/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection\"\nCSV_PATH = f\"{INPUT_DIR}/train.csv\"\n\nNEG_RATIO = 0.1  # 30%\n\nrandom.seed(42)\n\n# =========================================================\n# LOAD DATA\n# =========================================================\ndf = pd.read_csv(CSV_PATH)\n\n# =========================================================\n# SPLIT POS / NEG\n# =========================================================\ndf_pos = df[df[\"class_id\"] != 14]\ndf_neg = df[df[\"class_id\"] == 14]\n\n# unique image ids\npos_imgs = set(df_pos[\"image_id\"].unique())\nneg_imgs = set(df_neg[\"image_id\"].unique())\n\n# =========================================================\n# SAMPLE 30% NEGATIVE IMAGES\n# =========================================================\nsampled_neg_imgs = set(\n    random.sample(\n        list(neg_imgs),\n        int(len(neg_imgs) * NEG_RATIO)\n    )\n)\n\n# =========================================================\n# FINAL IMAGE SET (IMPORTANT: IMAGE LEVEL)\n# =========================================================\nfinal_imgs = pos_imgs | sampled_neg_imgs\n\n# =========================================================\n# STATS\n# =========================================================\nprint(\"===== VINBIGDATA SPLIT STATS =====\")\nprint(f\"Total images in dataset     : {len(pos_imgs | neg_imgs)}\")\nprint(f\"Positive images (any disease): {len(pos_imgs)}\")\nprint(f\"Negative images total       : {len(neg_imgs)}\")\n\nprint(\"\\n--- AFTER 30% SAMPLING ---\")\nprint(f\"Negative kept (30%)         : {len(sampled_neg_imgs)}\")\nprint(f\"Final images total          : {len(final_imgs)}\")\n\nprint(\"\\n--- RATIO ---\")\nprint(f\"Negative ratio in dataset    : {len(neg_imgs)/len(pos_imgs | neg_imgs):.2%}\")\nprint(f\"Negative ratio after split   : {len(sampled_neg_imgs)/len(final_imgs):.2%}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T08:29:35.278195Z","iopub.execute_input":"2026-05-12T08:29:35.27858Z","iopub.status.idle":"2026-05-12T08:29:35.425682Z","shell.execute_reply.started":"2026-05-12T08:29:35.278549Z","shell.execute_reply":"2026-05-12T08:29:35.424691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}