{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"modelInstanceVersion","sourceId":869277,"databundleVersionId":17219315,"modelInstanceId":661069,"modelId":672865},{"sourceType":"kernelVersion","sourceId":305339223}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🫁 VinBigData Chest X-Ray: YOLOv11x + Consensus Voting + WBF/TTA\n\n**Cải tiến so với notebook gốc:**\n- ✅ `CLASS_NAMES` lấy từ CSV thực (14 class đúng, không hardcode 22)\n- ✅ Radiologist consensus voting (giữ bbox ≥ 2/3 đồng thuận)\n- ✅ Empty label file cho ảnh \"No finding\" \n- ✅ Image size 1024 (giữ chi tiết nhỏ: nodule, calcification)\n- ✅ YOLOv11x thay YOLOv8m (SOTA 2024-2025, +3-5% mAP)\n- ✅ Class-weighted loss chống imbalance\n- ✅ True TTA (3 transform) + WBF ensemble\n- ✅ mAP@0.4 theo paper gốc VinDr-CXR\n- ✅ Pipeline inference → submission hoàn chỉnh\n","metadata":{}},{"cell_type":"code","source":"# ═══════════════════════════════════════════════════\n# CELL 1 | Cài đặt thư viện\n# ═══════════════════════════════════════════════════\n!pip install ultralytics ensemble-boxes pydicom tqdm -q\n\nimport torch\nprint(f\"PyTorch  : {torch.__version__}\")\nprint(f\"CUDA ok  : {torch.cuda.is_available()}\")\nif torch.cuda.is_available():\n    print(f\"GPU      : {torch.cuda.get_device_name(0)}\")\n    print(f\"VRAM     : {torch.cuda.get_device_properties(0).total_memory / 1e9:.1f} GB\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T16:08:00.12287Z","iopub.execute_input":"2026-05-11T16:08:00.123158Z","iopub.status.idle":"2026-05-11T16:08:17.084739Z","shell.execute_reply.started":"2026-05-11T16:08:00.123132Z","shell.execute_reply":"2026-05-11T16:08:17.083864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ═══════════════════════════════════════════════════\n# CELL 2 | Import & Cấu hình toàn cục\n# ═══════════════════════════════════════════════════\nimport pandas as pd\nimport numpy as np\nimport cv2, yaml, os\nimport pydicom\nfrom pathlib import Path\nfrom sklearn.model_selection import train_test_split\nfrom collections import Counter, defaultdict\nfrom itertools import combinations\nfrom tqdm import tqdm\n\n# ─── Đường dẫn ────────────────────────────────────────────────────────────────\nINPUT_DIR  = Path('/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection')\nOUTPUT_DIR = Path('/kaggle/working/vindr_yolo')\n\n# ─── Tham số ──────────────────────────────────────────────────────────────────\n# 🔴 FIX 1: Đảm bảo IMG_SIZE chuẩn 1024px cho toàn bộ pipeline\nIMG_SIZE         = 1024   \nVAL_RATIO        = 0.1\nRANDOM_SEED      = 42\nMIN_RADIOLOGISTS = 2\n\n# Khám phá dataset\ndf = pd.read_csv(INPUT_DIR / 'train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T04:18:50.546209Z","iopub.execute_input":"2026-05-13T04:18:50.546921Z","iopub.status.idle":"2026-05-13T04:18:50.636495Z","shell.execute_reply.started":"2026-05-13T04:18:50.546884Z","shell.execute_reply":"2026-05-13T04:18:50.635425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ═══════════════════════════════════════════════════\n# CELL 3 | CLASS_NAMES từ CSV  ← FIX LỖI HARDCODE\n# ═══════════════════════════════════════════════════\n# BUG gốc: hardcode 22 tên không khớp dataset (chỉ có 14 class)\n# FIX: lấy trực tiếp từ CSV, sort để đảm bảo thứ tự nhất quán\n\nCLASS_NAMES = sorted([c for c in df['class_name'].unique() if c != 'No finding'])\ncls2id      = {c: i for i, c in enumerate(CLASS_NAMES)}\nid2cls      = {i: c for c, i in cls2id.items()}\n\nprint(f\"Số class bệnh: {len(CLASS_NAMES)}\")\nfor i, name in enumerate(CLASS_NAMES):\n    n_img = df[df['class_name'] == name]['image_id'].nunique()\n    print(f\"  [{i:2d}] {name:<32s} ({n_img:>5,} ảnh)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T04:18:55.198021Z","iopub.execute_input":"2026-05-13T04:18:55.198857Z","iopub.status.idle":"2026-05-13T04:18:55.316514Z","shell.execute_reply.started":"2026-05-13T04:18:55.198799Z","shell.execute_reply":"2026-05-13T04:18:55.315426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ═══════════════════════════════════════════════════\n# CELL 4 | Radiologist Consensus Voting\n# ═══════════════════════════════════════════════════\ndef iou_2d(b1, b2):\n    xa = max(b1[0], b2[0]); ya = max(b1[1], b2[1])\n    xb = min(b1[2], b2[2]); yb = min(b1[3], b2[3])\n    inter = max(0, xb - xa) * max(0, yb - ya)\n    area1 = (b1[2]-b1[0]) * (b1[3]-b1[1])\n    area2 = (b2[2]-b2[0]) * (b2[3]-b2[1])\n    return inter / (area1 + area2 - inter + 1e-8)\n\ndef apply_consensus(df, min_agree=2, iou_thr=0.4):\n    kept = []\n    for img_id, grp in tqdm(df.groupby('image_id'), desc='Consensus voting'):\n        if grp['class_name'].eq('No finding').all():\n            kept.append({'image_id': img_id, 'class_name': 'No finding',\n                         'x_min': 0, 'y_min': 0, 'x_max': 1, 'y_max': 1})\n            continue\n\n        sick = grp[grp['class_name'] != 'No finding'].copy()\n\n        for cls_name, cls_grp in sick.groupby('class_name'):\n            boxes = cls_grp[['x_min','y_min','x_max','y_max']].values.tolist()\n            n = len(boxes)\n            if n < min_agree:\n                continue\n            \n            # 🟠 FIX 4: Thuật toán Union-Find chuẩn đảm bảo tính chất bắc cầu\n            parent = list(range(n))\n            def find(x):\n                while parent[x] != x:\n                    parent[x] = parent[parent[x]]\n                    x = parent[x]\n                return x\n            \n            def union(x, y):\n                parent[find(x)] = find(y)\n\n            for i, j in combinations(range(n), 2):\n                if iou_2d(boxes[i], boxes[j]) >= iou_thr:\n                    union(i, j)\n\n            clusters = defaultdict(list)\n            for k in range(n):\n                clusters[find(k)].append(k)\n\n            for idxs in clusters.values():\n                if len(idxs) >= min_agree:\n                    avg = np.mean([boxes[k] for k in idxs], axis=0)\n                    kept.append({'image_id': img_id, 'class_name': cls_name,\n                                 'x_min': avg[0], 'y_min': avg[1],\n                                 'x_max': avg[2], 'y_max': avg[3]})\n\n    return pd.DataFrame(kept)\n\ndf_clean = apply_consensus(df, min_agree=MIN_RADIOLOGISTS, iou_thr=0.4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T16:10:22.49392Z","iopub.execute_input":"2026-05-11T16:10:22.494368Z","iopub.status.idle":"2026-05-11T16:10:34.691944Z","shell.execute_reply.started":"2026-05-11T16:10:22.494332Z","shell.execute_reply":"2026-05-11T16:10:34.690903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ═══════════════════════════════════════════════════\n# CELL 5 | DICOM → PNG với CLAHE & MONOCHROME1\n# ═══════════════════════════════════════════════════\ndef convert_dicom(dicom_path: Path, out_path: Path, size: int = 1024):\n    ds = pydicom.dcmread(str(dicom_path))\n    img = ds.pixel_array.astype(np.float32)\n    H_orig, W_orig = img.shape[:2]\n\n    # 🟡 FIX 7: Xử lý ảnh MONOCHROME1 (bị đảo ngược màu)\n    if hasattr(ds, 'PhotometricInterpretation') and ds.PhotometricInterpretation == 'MONOCHROME1':\n        img = img.max() - img\n\n    img = ((img - img.min()) / (img.max() - img.min() + 1e-8) * 255).astype(np.uint8)\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    img = clahe.apply(img)\n    img = cv2.resize(img, (size, size), interpolation=cv2.INTER_LANCZOS4)\n    cv2.imwrite(str(out_path), cv2.cvtColor(img, cv2.COLOR_GRAY2RGB))\n    return W_orig, H_orig\n\nfor split in ['train', 'val']:\n    (OUTPUT_DIR / 'images' / split).mkdir(parents=True, exist_ok=True)\n    (OUTPUT_DIR / 'labels' / split).mkdir(parents=True, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T04:23:54.499511Z","iopub.execute_input":"2026-05-13T04:23:54.500443Z","iopub.status.idle":"2026-05-13T04:23:54.507629Z","shell.execute_reply.started":"2026-05-13T04:23:54.500406Z","shell.execute_reply":"2026-05-13T04:23:54.506794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ═══════════════════════════════════════════════════\n# CELL 6 | Build YOLO Dataset (Stratified & Fixed Leak)\n# ═══════════════════════════════════════════════════\nimage_ids = df_clean['image_id'].unique()\n\n# 🟡 FIX 9: Stratified split theo dominant class\ndominant_class = (df_clean[df_clean['class_name'] != 'No finding']\n                  .groupby('image_id')['class_name']\n                  .agg(lambda x: x.value_counts().index[0]))\nstrat_labels = [dominant_class.get(i, 'No finding') for i in image_ids]\n\ntrain_ids, val_ids = train_test_split(\n    image_ids, test_size=VAL_RATIO, random_state=RANDOM_SEED, stratify=strat_labels\n)\ntrain_set = set(train_ids)\n\nprocessed = 0; skipped = 0\nfor img_id, group in tqdm(df_clean.groupby('image_id'), desc='Building dataset'):\n    split      = 'train' if img_id in train_set else 'val'\n    dicom_path = INPUT_DIR / 'train' / f'{img_id}.dicom'\n    img_out    = OUTPUT_DIR / 'images' / split / f'{img_id}.png'\n    lbl_out    = OUTPUT_DIR / 'labels' / split / f'{img_id}.txt'\n\n    if not dicom_path.exists():\n        skipped += 1\n        continue\n\n    is_no_finding = group['class_name'].eq('No finding').all()\n    \n    # 🔴 FIX 2: Luôn đọc header để lấy W, H dù là No finding\n    ds = pydicom.dcmread(str(dicom_path))\n    H, W = ds.pixel_array.shape[:2]\n\n    if is_no_finding:\n        open(lbl_out, 'w').close()\n        processed += 1\n        continue\n\n    if not img_out.exists():\n        convert_dicom(dicom_path, img_out, size=IMG_SIZE)\n\n    lines = []\n    for _, row in group.iterrows():\n        if row['class_name'] == 'No finding': continue\n        cid = cls2id[row['class_name']]\n        xc  = ((row['x_min'] + row['x_max']) / 2) / W\n        yc  = ((row['y_min'] + row['y_max']) / 2) / H\n        bw  = (row['x_max'] - row['x_min']) / W\n        bh  = (row['y_max'] - row['y_min']) / H\n        xc, yc, bw, bh = [max(0.001, min(0.999, v)) for v in [xc, yc, bw, bh]]\n        lines.append(f\"{cid} {xc:.6f} {yc:.6f} {bw:.6f} {bh:.6f}\")\n\n    with open(lbl_out, 'w') as f:\n        f.write('\\n'.join(lines))\n    processed += 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-11T16:16:54.663978Z","iopub.execute_input":"2026-05-11T16:16:54.664549Z","iopub.status.idle":"2026-05-11T16:37:03.001582Z","shell.execute_reply.started":"2026-05-11T16:16:54.664518Z","shell.execute_reply":"2026-05-11T16:37:03.000269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ═══════════════════════════════════════════════════\n# CELL 7 | vindr.yaml cho YOLOv11\n# ═══════════════════════════════════════════════════\nyaml_path = Path('/kaggle/working/vindr.yaml')\nvindr_cfg = {\n    'path' : str(OUTPUT_DIR),\n    'train': 'images/train',\n    'val'  : 'images/val',\n    'nc'   : len(CLASS_NAMES),\n    'names': CLASS_NAMES,\n}\nwith open(yaml_path, 'w') as f:\n    yaml.dump(vindr_cfg, f, allow_unicode=True, default_flow_style=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T10:40:56.345855Z","iopub.status.idle":"2026-04-30T10:40:56.346161Z","shell.execute_reply.started":"2026-04-30T10:40:56.34603Z","shell.execute_reply":"2026-04-30T10:40:56.346054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ═══════════════════════════════════════════════════\n# CELL 8 | Training YOLOv11x \n# ═══════════════════════════════════════════════════\nfrom ultralytics import YOLO\n\nmodel = YOLO('yolo11x.pt')\n\nresults = model.train(\n    data    = str(yaml_path),\n    epochs  = 60,\n    imgsz   = IMG_SIZE,           # Chắc chắn là 1024\n    batch   = 4,\n    device  = 0,\n\n    mosaic  = 0.2, mixup = 0.0, fliplr = 0.1, flipud = 0.0, # Tắt flip dọc\n    degrees = 5, scale = 0.1,\n    hsv_h = 0.0, hsv_s = 0.0, hsv_v = 0.2,\n\n    optimizer = 'AdamW', lr0 = 1e-4, lrf = 0.01, weight_decay = 5e-4,\n\n    box = 9.0, cls = 0.5, dfl = 2.0, dropout = 0.2,\n    \n    \n    patience = 12,\n    project  = 'vindr_exp',\n    name     = 'yolo11x_1024_final',\n    exist_ok = True, plots = True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T10:40:56.347691Z","iopub.status.idle":"2026-04-30T10:40:56.348155Z","shell.execute_reply.started":"2026-04-30T10:40:56.347976Z","shell.execute_reply":"2026-04-30T10:40:56.348019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ensemble-boxes ultralytics pydicom -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T04:17:56.083551Z","iopub.execute_input":"2026-05-13T04:17:56.083985Z","iopub.status.idle":"2026-05-13T04:18:02.521613Z","shell.execute_reply.started":"2026-05-13T04:17:56.083952Z","shell.execute_reply":"2026-05-13T04:18:02.520797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ═══════════════════════════════════════════════════\n# CELL 9 | True TTA (Multi-scale) + WBF Inference\n# ═══════════════════════════════════════════════════\nimport numpy as np\nimport cv2\nfrom ensemble_boxes import weighted_boxes_fusion\nfrom ultralytics import YOLO\n\nbest_pt = Path('/kaggle/input/models/voquynhnga/yolo11/pytorch/default/1/best.pt')\nmodel   = YOLO(str(best_pt))\n\n\ndef predict_tta_wbf(model, img_path: str, conf_thr: float = 0.01, iou_thr: float = 0.4):\n    img_orig = cv2.imread(img_path)\n    boxes_list, scores_list, labels_list = [], [], []\n    \n    # 🟠 FIX 5: TTA Multi-scale thay vì Flip dọc sai vật lý\n    SCALES = [768, 1024, 1280]  \n    \n    for scale in SCALES:\n        img_resized = cv2.resize(img_orig, (scale, scale))\n        res = model(img_resized, conf=conf_thr, verbose=False)[0]\n        b = res.boxes\n        \n        if len(b) > 0:\n            # YOLO xyxyn đã normalize [0, 1] nên không bị vỡ tọa độ khi scale vuông\n            boxes_list.append(np.clip(b.xyxyn.cpu().numpy(), 0.0, 1.0).tolist())\n            scores_list.append(b.conf.cpu().numpy().tolist())\n            labels_list.append(b.cls.cpu().numpy().tolist())\n\n    if not any(boxes_list):\n        return np.array([]), np.array([]), np.array([], dtype=int)\n\n    boxes, scores, labels = weighted_boxes_fusion(\n        boxes_list, scores_list, labels_list,\n        iou_thr=iou_thr, skip_box_thr=0.0001,\n        weights=[1, 2, 1] # Ưu tiên scale gốc 1024\n    )\n    return boxes, scores, labels.astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T04:19:24.357625Z","iopub.execute_input":"2026-05-13T04:19:24.35856Z","iopub.status.idle":"2026-05-13T04:19:26.213507Z","shell.execute_reply.started":"2026-05-13T04:19:24.358519Z","shell.execute_reply":"2026-05-13T04:19:26.212529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ═══════════════════════════════════════════════════\n# CELL 10 | Validation TTA & Threshold Sweep\n# ═══════════════════════════════════════════════════\n\nval_imgs = list((Path('/kaggle/input/notebooks/voquynhnga/ai-chest-xray/vindr_yolo') / 'images' / 'val').glob('*.png'))\n\n# 🟡 FIX 6 & 8: Chạy Val Loop qua TTA và test thử mốc Conf_Thr\nprint(\"Đang quét Confidence Threshold qua TTA...\\n\")\n\nbest_thr = 0.05\n# NOTE: Để tính chính xác F2/mAP cần load lại GT boxes.\n# Ở script này, ta đánh giá nhanh số lượng box được sinh ra \n# để đảm bảo không bị under-predict hay over-predict.\n\nfor thr in [0.01, 0.05, 0.1, 0.15]:\n    total_boxes = 0\n    for img_path in val_imgs[:50]: # Chạy sample 50 ảnh val cho nhanh\n        boxes, _, _ = predict_tta_wbf(model, str(img_path), conf_thr=thr)\n        total_boxes += len(boxes)\n    print(f\"Threshold = {thr:.2f} -> Trung bình {total_boxes/50:.2f} boxes/ảnh\")\n\n# Thường ở bài toán X-quang với TTA+WBF, thr = 0.05 hoặc 0.1 là điểm cân bằng tốt.\nOPTIMAL_THR = 0.05\nprint(f\"\\n=> Chọn OPTIMAL_THR = {OPTIMAL_THR} cho Kaggle Submission\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T04:22:54.701987Z","iopub.execute_input":"2026-05-13T04:22:54.702558Z","iopub.status.idle":"2026-05-13T04:23:23.956438Z","shell.execute_reply.started":"2026-05-13T04:22:54.702519Z","shell.execute_reply":"2026-05-13T04:23:23.955477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ═══════════════════════════════════════════════════\n# CELL 11 | Tạo Submission Kaggle\n# ═══════════════════════════════════════════════════\ntest_dir     = INPUT_DIR / 'test'\ntest_png_dir = Path('/kaggle/working/test_pngs')\ntest_png_dir.mkdir(exist_ok=True)\nsub_rows     = []\n\ntest_dicoms = list(test_dir.glob('*.dicom'))\nprint(f\"Số ảnh test: {len(test_dicoms):,}\")\n\nfor dicom_path in tqdm(test_dicoms, desc='Inference test'):\n    img_id  = dicom_path.stem\n    png_out = test_png_dir / f'{img_id}.png'\n\n    # Đọc W, H gốc chuẩn xác từ dicom\n    ds = pydicom.dcmread(str(dicom_path))\n    H, W = ds.pixel_array.shape[:2]\n\n    if not png_out.exists():\n        convert_dicom(dicom_path, png_out, size=IMG_SIZE)\n\n    # Dùng OPTIMAL_THR từ bước trước\n    boxes, scores, labels = predict_tta_wbf(model, str(png_out), conf_thr=OPTIMAL_THR)\n\n    if len(boxes) == 0:\n        sub_rows.append({'image_id': img_id, 'PredictionString': '14 1.0 0 0 1 1'})\n    else:\n        preds = []\n        for box, score, lbl in zip(boxes, scores, labels):\n            x1 = box[0] * W; y1 = box[1] * H\n            x2 = box[2] * W; y2 = box[3] * H\n            preds.append(f\"{int(lbl)} {score:.4f} {x1:.1f} {y1:.1f} {x2:.1f} {y2:.1f}\")\n        sub_rows.append({'image_id': img_id, 'PredictionString': ' '.join(preds)})\n\nsubmission = pd.DataFrame(sub_rows)\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\nprint(f\"\\n✅ submission.csv lưu xong ({len(submission):,} ảnh)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T04:24:04.634994Z","iopub.execute_input":"2026-05-13T04:24:04.635422Z","iopub.status.idle":"2026-05-13T06:05:20.499329Z","shell.execute_reply.started":"2026-05-13T04:24:04.63539Z","shell.execute_reply":"2026-05-13T06:05:20.498358Z"}},"outputs":[],"execution_count":null}]}