{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q pydicom\n!pip install -q python-gdcm # Cần thiết để giải mã một số chuẩn nén DICOM phức tạp\n!pip install -q ensemble-boxes","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-12T02:41:09.168499Z","iopub.execute_input":"2026-05-12T02:41:09.168862Z","iopub.status.idle":"2026-05-12T02:41:20.547792Z","shell.execute_reply.started":"2026-05-12T02:41:09.168832Z","shell.execute_reply":"2026-05-12T02:41:20.546671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!rm -rf /kaggle/working/yolo_data\n!rm -rf /kaggle/working/train_images_1024","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T02:41:20.550237Z","iopub.execute_input":"2026-05-12T02:41:20.550546Z","iopub.status.idle":"2026-05-12T02:41:20.787076Z","shell.execute_reply.started":"2026-05-12T02:41:20.550511Z","shell.execute_reply":"2026-05-12T02:41:20.78604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom tqdm.auto import tqdm\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    dicom = pydicom.dcmread(path)\n    \n    if voi_lut:\n        try:\n            data = apply_voi_lut(dicom.pixel_array, dicom)\n        except:\n            data = dicom.pixel_array\n    else:\n        data = dicom.pixel_array\n               \n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    \n    # --- CLAHE---\n    clahe = cv2.createCLAHE(clipLimit=2.5, tileGridSize=(8, 8))\n    data = clahe.apply(data)\n    # -------------------------------------------------\n    \n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T02:41:20.788623Z","iopub.execute_input":"2026-05-12T02:41:20.788962Z","iopub.status.idle":"2026-05-12T02:41:20.796646Z","shell.execute_reply.started":"2026-05-12T02:41:20.788928Z","shell.execute_reply":"2026-05-12T02:41:20.795645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Khai báo đường dẫn từ dữ liệu gốc của bạn\nRAW_TRAIN_DIR = '/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection/train'\nOUTPUT_DIR = '/kaggle/working/train_images_1024'\nTARGET_SIZE = 1024 # Resize xuống 1024x1024 để tránh tràn ổ cứng Kaggle\n\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\n# # Lấy danh sách file DICOM\ndicom_files = [f for f in os.listdir(RAW_TRAIN_DIR) if f.endswith('.dicom')]\n\n# # Giới hạn số lượng chạy thử (bạn có thể bỏ [0:3000]] để chạy toàn bộ khi đã chắc chắn)\n# for filename in tqdm(dicom_files, desc=\"Converting DICOM to PNG\"):\n#     dicom_path = os.path.join(RAW_TRAIN_DIR, filename)\n    \n#     try:\n#         # 1. Đọc và chuẩn hóa ảnh\n#         img = read_xray(dicom_path)\n        \n#         # 2. Resize ảnh để giảm dung lượng\n#         img_resized = cv2.resize(img, (TARGET_SIZE, TARGET_SIZE))\n        \n#         # 3. Lưu ảnh dưới dạng JPG\n#         output_filename = filename.replace('.dicom', '.png')\n#         output_path = os.path.join(OUTPUT_DIR, output_filename)\n#         cv2.imwrite(output_path, img_resized)\n        \n#     except Exception as e:\n#         print(f\"Lỗi khi xử lý file {filename}: {e}\")\n\n# print(\"Quá trình chuyển đổi hoàn tất!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T02:41:20.797772Z","iopub.execute_input":"2026-05-12T02:41:20.79806Z","iopub.status.idle":"2026-05-12T02:41:21.178754Z","shell.execute_reply.started":"2026-05-12T02:41:20.798033Z","shell.execute_reply":"2026-05-12T02:41:21.177986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# [Cell 3 - Cập nhật] Resize ảnh giữ nguyên tỷ lệ (Letterbox)\ndef resize_with_padding(img, target_size=1024):\n    old_size = img.shape[:2] # (height, width)\n    ratio = float(target_size) / max(old_size)\n    new_size = tuple([int(x * ratio) for x in old_size])\n\n    # Resize ảnh theo tỷ lệ cạnh dài nhất\n    img_resized = cv2.resize(img, (new_size[1], new_size[0]))\n\n    # Tạo canvas đen 1024x1024\n    delta_w = target_size - new_size[1]\n    delta_h = target_size - new_size[0]\n    top, bottom = delta_h // 2, delta_h - (delta_h // 2)\n    left, right = delta_w // 2, delta_w - (delta_w // 2)\n\n    color = [0, 0, 0]\n    new_img = cv2.copyMakeBorder(img_resized, top, bottom, left, right, cv2.BORDER_CONSTANT, value=color)\n    return new_img, ratio, top, left\n\n# Chạy vòng lặp chuyển đổi\nfor filename in tqdm(dicom_files, desc=\"Converting DICOM with Letterbox\"):\n    dicom_path = os.path.join(RAW_TRAIN_DIR, filename)\n    try:\n        img = read_xray(dicom_path)\n        # Sử dụng hàm resize mới\n        img_padded, scale_ratio, pad_top, pad_left = resize_with_padding(img, TARGET_SIZE)\n        \n        # Sửa .jpg thành .png\n        output_filename = filename.replace('.dicom', '.png')\n        cv2.imwrite(os.path.join(OUTPUT_DIR, output_filename), img_padded)\n    except Exception as e:\n        print(f\"Lỗi: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T02:41:21.181168Z","iopub.execute_input":"2026-05-12T02:41:21.181494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# [Cell 4] Lấy kích thước ảnh gốc siêu tốc từ DICOM Header\nimport pydicom\nfrom tqdm.auto import tqdm\nimport pandas as pd\nimport os\n\nRAW_TRAIN_DIR = '/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection/train'\ndicom_files = [f for f in os.listdir(RAW_TRAIN_DIR) if f.endswith('.dicom')]\n\nmeta_data = []\n\n# Quét qua toàn bộ file (Tốc độ sẽ rất nhanh vì không tải ảnh pixel)\nfor filename in tqdm(dicom_files, desc=\"Đọc DICOM Headers\"):\n    dicom_path = os.path.join(RAW_TRAIN_DIR, filename)\n    image_id = filename.replace('.dicom', '')\n    \n    try:\n        # stop_before_pixels=True giúp bỏ qua phần ảnh nặng nề, chỉ đọc text thông tin\n        dicom = pydicom.dcmread(dicom_path, stop_before_pixels=True)\n        width = dicom.Columns\n        height = dicom.Rows\n        meta_data.append({\n            'image_id': image_id,\n            'orig_width': width,\n            'orig_height': height\n        })\n    except Exception as e:\n        print(f\"Lỗi đọc metadata {filename}: {e}\")\n\n# Lưu thành DataFrame\ndf_meta = pd.DataFrame(meta_data)\nprint(\"Hoàn tất lấy kích thước gốc! Số lượng:\", len(df_meta))\ndf_meta.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# [Cell 5 - ĐÃ FIX LỖI NaN] Tính lại tọa độ theo ảnh đã Padding\nPATH_TO_TRAIN_CSV = '/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection/train.csv'\ndf_train = pd.read_csv(PATH_TO_TRAIN_CSV)\n\n# Merge metadata vào df_train\ndf_train = df_train.merge(df_meta, on='image_id', how='left')\n\n# ĐIỂM SỬA QUAN TRỌNG: Chỉ giữ lại các ảnh đã có thông tin meta (đã chạy qua Cell 4)\n# Nếu bạn chỉ chạy thử 30 ảnh ở Cell 4, dòng này sẽ lọc ra đúng 30 ảnh đó để xử lý tiếp\ndf_train = df_train.dropna(subset=['orig_width', 'orig_height'])\n\n# Lọc các ảnh có bệnh\ndf_disease = df_train[df_train['class_id'] != 14].copy()\nTARGET_SIZE = 1024\n\ndef update_box_coords(row):\n    w, h = row['orig_width'], row['orig_height']\n    \n    # Tính scale và padding\n    scale = TARGET_SIZE / max(w, h)\n    new_w, new_h = int(w * scale), int(h * scale)\n    \n    pad_left = (TARGET_SIZE - new_w) // 2\n    pad_top = (TARGET_SIZE - new_h) // 2\n    \n    # Cập nhật tọa độ mới\n    row['x_min'] = row['x_min'] * scale + pad_left\n    row['x_max'] = row['x_max'] * scale + pad_left\n    row['y_min'] = row['y_min'] * scale + pad_top\n    row['y_max'] = row['y_max'] * scale + pad_top\n    return row\n\n# Áp dụng hàm cập nhật\nif not df_disease.empty:\n    df_disease = df_disease.apply(update_box_coords, axis=1)\n    \n    # Lưu CSV mới\n    df_disease[['image_id', 'class_id', 'rad_id', 'x_min', 'y_min', 'x_max', 'y_max']].to_csv('/kaggle/working/train_1024.csv', index=False)\n    print(f\"✅ Đã cập nhật tọa độ cho {df_disease['image_id'].nunique()} ảnh!\")\nelse:\n    print(\"⚠️ Cảnh báo: Không có ảnh có bệnh nào được tìm thấy trong tập metadata hiện tại.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# [Cell 6 - FIX LỆCH PHA DỮ LIỆU] Kiểm tra trực quan ảnh và box đã Padding\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport cv2\nimport pandas as pd\nimport os\n\nnew_csv_path = '/kaggle/working/train_1024.csv'\ndf_new = pd.read_csv(new_csv_path)\n\n# 1. Lấy danh sách ID ảnh THỰC TẾ đang nằm trong folder\nimg_dir = '/kaggle/working/train_images_1024'\navailable_imgs = [f.replace('.png', '') for f in os.listdir(img_dir) if f.endswith('.png')]\n\n# 2. Lọc file CSV: Chỉ giữ lại tọa độ của những ảnh đã có file .jpg\ndf_available = df_new[df_new['image_id'].isin(available_imgs)]\n\nif not df_available.empty:\n    # Lấy ID của bức ảnh hợp lệ đầu tiên\n    sample_img_id = df_available['image_id'].iloc[0]\n    print(f\"Đang hiển thị thử ảnh: {sample_img_id}\")\n    \n    padded_img_path = f'{img_dir}/{sample_img_id}.png'\n    boxes = df_available[df_available['image_id'] == sample_img_id]\n\n    img = cv2.imread(padded_img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    fig, ax = plt.subplots(figsize=(10, 10))\n    ax.imshow(img)\n    ax.axis('off')\n\n    for _, row in boxes.iterrows():\n        rect = patches.Rectangle((row['x_min'], row['y_min']),\n                                 row['x_max'] - row['x_min'],\n                                 row['y_max'] - row['y_min'],\n                                 linewidth=3, edgecolor='r', facecolor='none', \n                                 label=f\"Class {int(row['class_id'])}\")\n        ax.add_patch(rect)\n\n    # Rút gọn chú thích\n    handles, labels = ax.get_legend_handles_labels()\n    by_label = dict(zip(labels, handles))\n    ax.legend(by_label.values(), by_label.keys(), loc='upper right')\n    \n    plt.title(f\"Kiểm tra Padding cho Image: {sample_img_id}\")\n    plt.show()\nelse:\n    print(\"❌ Trong số các ảnh bạn vừa convert, không có ảnh nào chứa mầm bệnh để hiển thị!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# [Cell 7 - CHUẨN THEO BÀI BÁO VINBIGDATA] Thực thi WBF gộp nhãn đa bác sĩ\nimport os\nimport pandas as pd\nimport numpy as np\nimport shutil\nfrom tqdm.auto import tqdm\nfrom ensemble_boxes import weighted_boxes_fusion\n\nWORK_DIR = '/kaggle/working/yolo_data'\nIMG_DIR = '/kaggle/working/train_images_1024' \nCSV_PATH = '/kaggle/working/train_1024.csv' \n\n# 1. Tạo cấu trúc thư mục\nfor s in ['train', 'val']:\n    os.makedirs(f'{WORK_DIR}/images/{s}', exist_ok=True)\n    os.makedirs(f'{WORK_DIR}/labels/{s}', exist_ok=True)\n\ndf = pd.read_csv(CSV_PATH)\nIMG_SIZE = 1024\n\n# Trọng số độ tin cậy của bác sĩ (Dùng làm weights cho WBF)\nrad_weights = {\n    'R8': 1.0, 'R9': 1.0, 'R10': 1.0,\n    'R1': 0.8, 'R2': 0.8, 'R3': 0.8, 'R4': 0.8, 'R5': 0.8, 'R6': 0.8, 'R7': 0.8,\n    'R11': 0.9, 'R12': 0.9, 'R13': 0.9, 'R14': 0.9, 'R15': 0.9, 'R16': 0.9, 'R17': 0.9\n}\n\nimage_ids = df['image_id'].unique()\n\n# 2. Chia Train/Val (80/20) để chống Data Leakage\nnp.random.seed(42)\nval_img_ids = set(np.random.choice(image_ids, int(len(image_ids) * 0.2), replace=False))\n\n# 3. Vòng lặp WBF nhóm theo từng bác sĩ\nfor img_id in tqdm(image_ids, desc=\"Gộp nhãn WBF đa bác sĩ...\"):\n    current_split = 'val' if img_id in val_img_ids else 'train'\n    img_df = df[df['image_id'] == img_id]\n    \n    boxes_list = []\n    scores_list = []\n    labels_list = []\n    weights_list = []\n    \n    # ĐIỂM SỬA LÕI: Phân tách box thuộc về từng bác sĩ riêng biệt\n    for rad_id, group in img_df.groupby('rad_id'):\n        # Tọa độ chuẩn hóa về [0, 1]\n        boxes = group[['x_min', 'y_min', 'x_max', 'y_max']].values / IMG_SIZE\n        boxes_list.append(boxes.tolist())\n        \n        # Nhãn (Class ID)\n        labels = group['class_id'].values\n        labels_list.append(labels.tolist())\n        \n        # Độ tin cậy mặc định lấy theo hệ số của bác sĩ\n        weight = rad_weights.get(rad_id, 1.0)\n        scores = [weight] * len(group)\n        \n        scores_list.append(scores)\n        weights_list.append(weight)\n\n    # Chạy WBF để hòa trộn box từ các list bác sĩ khác nhau\n    boxes_wbf, scores_wbf, labels_wbf = weighted_boxes_fusion(\n        boxes_list, scores_list, labels_list, \n        weights=weights_list, iou_thr=0.4, skip_box_thr=0.001\n    )\n\n    # 4. Chuyển sang định dạng YOLO (cx, cy, w, h)\n    yolo_boxes = []\n    for box, label in zip(boxes_wbf, labels_wbf):\n        x_min, y_min, x_max, y_max = box\n        \n        # Cắt gọt tọa độ tránh tràn viền do sai số thập phân\n        x_min, y_min = max(0.0, x_min), max(0.0, y_min)\n        x_max, y_max = min(1.0, x_max), min(1.0, y_max)\n        \n        cx = (x_min + x_max) / 2.0\n        cy = (y_min + y_max) / 2.0\n        w = x_max - x_min\n        h = y_max - y_min\n        \n        yolo_boxes.append(f\"{int(label)} {cx:.6f} {cy:.6f} {w:.6f} {h:.6f}\")\n\n    # Ghi file label và copy ảnh\n    label_path = f\"{WORK_DIR}/labels/{current_split}/{img_id}.txt\"\n    with open(label_path, \"w\") as f:\n        f.write(\"\\n\".join(yolo_boxes))\n\n    src_img = f\"{IMG_DIR}/{img_id}.png\"\n    dst_img = f\"{WORK_DIR}/images/{current_split}/{img_id}.png\"\n    if os.path.exists(src_img) and not os.path.exists(dst_img):\n        shutil.copy(src_img, dst_img)\n\nprint(f\"✅ Hoàn tất! Dữ liệu WBF chuẩn bài báo nằm tại: {WORK_DIR}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# [Cell 8 - ĐÃ FIX LỖI] Bổ sung ảnh Background (Class 14) vào tập YOLO Train\nimport random\nimport os\nimport shutil\nimport pandas as pd\nfrom tqdm.auto import tqdm\n\n# 1. Khai báo đường dẫn đồng bộ với các cell trước\nWORK_DIR = '/kaggle/working/yolo_data'\nIMG_DIR = '/kaggle/working/train_images_1024'\nORIG_CSV_PATH = '/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection/train.csv'\nDISEASE_CSV_PATH = '/kaggle/working/train_1024.csv'\n\n# 2. Đọc file CSV gốc để tìm các ảnh \"No finding\" (Class 14)\ndf_orig = pd.read_csv(ORIG_CSV_PATH)\n# Lấy danh sách ID các ảnh chỉ có nhãn 14 (bình thường)\nall_bg_ids = df_orig[df_orig['class_id'] == 14]['image_id'].unique().tolist()\n\n# 3. Kiểm tra xem trong số 30 ảnh bạn vừa chạy thử (Cell 3), có ảnh nào là background không\n# (Bước này quan trọng vì bạn đang chạy mode debug 30 ảnh)\navailable_processed_ids = [f.replace('.png', '') for f in os.listdir(IMG_DIR)]\nbg_ids_available = [img_id for img_id in all_bg_ids if img_id in available_processed_ids]\n\n# 4. Đọc file CSV bệnh để tính toán tỷ lệ 15%\ndf_disease = pd.read_csv(DISEASE_CSV_PATH)\nnum_disease_images = df_disease['image_id'].nunique()\nnum_bg_to_add = int(num_disease_images * 0.15)\n\n# Đảm bảo không lấy quá số lượng background đang có sẵn trong folder images\nnum_bg_to_add = min(len(bg_ids_available), num_bg_to_add)\n\nif num_bg_to_add > 0:\n    random.seed(42)\n    selected_bg_images = random.sample(bg_ids_available, num_bg_to_add)\n\n    # 5. Copy ảnh background vào folder YOLO và tạo file label rỗng\n    split = 'train'\n    for img_id in tqdm(selected_bg_images, desc=f\"Thêm ảnh background vào {split}\"):\n        src_img = f\"{IMG_DIR}/{img_id}.png\"\n        dst_img = f\"{WORK_DIR}/images/{split}/{img_id}.png\"\n        \n        if os.path.exists(src_img) and not os.path.exists(dst_img):\n            shutil.copy(src_img, dst_img)\n\n        # Tạo file label .txt rỗng (0 bytes) - YOLO hiểu đây là ảnh không có vật thể\n        txt_path = f\"{WORK_DIR}/labels/{split}/{img_id}.txt\"\n        with open(txt_path, \"w\") as f:\n            pass\n\n    print(f\"✅ Hoàn tất! Đã thêm {len(selected_bg_images)} ảnh background vào tập YOLO {split}.\")\nelse:\n    print(\"⚠️ Không tìm thấy ảnh background nào trong danh sách 30 ảnh chạy thử.\")\n    print(\"Mẹo: Khi bạn chạy toàn bộ dữ liệu (bỏ [0:30]), cell này sẽ tự động hoạt động chính xác.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# [Cell Cuối] Thống kê toàn bộ cấu trúc thư mục và số lượng file\nimport os\n\n# Đường dẫn gốc từ input bạn đã add\nbase_path = '/kaggle/input/notebooks/nguyenphanthanhan/notebooka9deacebf8'\n# Đường dẫn dữ liệu bạn vừa tiền xử lý xong\nworking_path = '/kaggle/working'\n\ndef count_files(path):\n    for root, dirs, files in os.walk(path):\n        if len(files) > 0:\n            print(f\"Thư mục: {root}\")\n            print(f\" - Có {len(files)} files.\")\n\nprint(\"--- THỐNG KÊ FILE TRONG INPUT ---\")\nif os.path.exists(base_path):\n    count_files(base_path)\nelse:\n    print(f\"⚠️ Không tìm thấy đường dẫn input: {base_path}\")\n\nprint(\"\\n--- THỐNG KÊ FILE ĐÃ TẠO TRONG WORKING ---\")\nif os.path.exists(working_path):\n    count_files(working_path)\nelse:\n    print(f\"⚠️ Không tìm thấy đường dẫn working: {working_path}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\n# 1. Đọc file CSV gốc chứa nhãn từ ban tổ chức\nORIG_CSV_PATH = '/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection/train.csv'\ndf_orig = pd.read_csv(ORIG_CSV_PATH)\n\n# 2. Lấy danh sách ID ảnh đã thực sự được xử lý và lưu vào folder (bỏ đuôi .jpg)\nprocessed_ids = [f.split('.')[0] for f in os.listdir('/kaggle/working/train_images_1024')]\n\n# 3. Lọc DataFrame gốc để chỉ giữ lại thông tin của các ảnh đã xử lý\ncheck_df = df_orig[df_orig['image_id'].isin(processed_ids)].copy()\n\n# 4. Phân loại nhị phân: Class 14 là 0 (Bình thường), các Class 0-13 là 1 (Có bệnh)\ncheck_df['target'] = check_df['class_id'].apply(lambda x: 0 if x == 14 else 1)\n\n# Vì một ảnh có thể được 3 bác sĩ gán nhiều nhãn (nhiều dòng trong CSV), \n# ta sẽ gộp lại theo từng image_id để đếm chính xác số lượng BỨC ẢNH thay vì số lượng BOX.\n# Lấy max() để nếu ảnh có bất kỳ nhãn bệnh nào (1) thì sẽ được tính là có bệnh.\nimg_distribution = check_df.groupby('image_id')['target'].max()\n\n# 5. In kết quả\nprint(f\"Tổng số ảnh trong thư mục: {len(processed_ids)}\")\nprint(\"-\" * 35)\nprint(\"Thống kê phân bố (0: Bình thường, 1: Có bệnh):\")\ndistribution_counts = img_distribution.value_counts().rename(index={0: '0 (Bình thường)', 1: '1 (Có bệnh)'})\nprint(distribution_counts)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}