{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":15892005,"datasetId":10190211,"databundleVersionId":16846413},{"sourceType":"datasetVersion","sourceId":15094918,"datasetId":9664561,"databundleVersionId":15979687},{"sourceType":"datasetVersion","sourceId":15588375,"datasetId":9973820,"databundleVersionId":16520657},{"sourceType":"datasetVersion","sourceId":16100879,"datasetId":10323651,"databundleVersionId":17071756}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":7665.964729,"end_time":"2026-04-25T08:58:07.058265","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-04-25T06:50:21.093536","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"d7fff5df","cell_type":"code","source":"!pip -q install pydicom pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg ultralytics albumentations\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2026-04-25T06:50:23.71412Z","iopub.status.busy":"2026-04-25T06:50:23.713466Z","iopub.status.idle":"2026-04-25T06:50:30.352946Z","shell.execute_reply":"2026-04-25T06:50:30.352198Z"},"papermill":{"duration":6.648491,"end_time":"2026-04-25T06:50:30.354764","exception":false,"start_time":"2026-04-25T06:50:23.706273","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"a5b92ac4","cell_type":"code","source":"import os\nimport cv2\nimport yaml\nimport time\nimport random\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport albumentations as A\n\nfrom collections import defaultdict\nfrom concurrent.futures import ThreadPoolExecutor, as_completed\n","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:50:30.365848Z","iopub.status.busy":"2026-04-25T06:50:30.365221Z","iopub.status.idle":"2026-04-25T06:50:38.415228Z","shell.execute_reply":"2026-04-25T06:50:38.414575Z"},"papermill":{"duration":8.057332,"end_time":"2026-04-25T06:50:38.416938","exception":false,"start_time":"2026-04-25T06:50:30.359606","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"ddb1b03e","cell_type":"markdown","source":"# Config","metadata":{"papermill":{"duration":0.004762,"end_time":"2026-04-25T06:50:38.426778","exception":false,"start_time":"2026-04-25T06:50:38.422016","status":"completed"},"tags":[]}},{"id":"858acb27","cell_type":"code","source":"# =========================\n# PATH CONFIG\n# =========================\nfrom pathlib import Path\n\nROOT = Path(\"/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection\")\n\nTRAIN_DIR = ROOT / \"train\"\nTEST_DIR = ROOT / \"test\"\n\n# Label CSV giống LUNG_YOLO_CODE/train.py: train+test đã merge sẵn, có khoảng 1000 No finding.\nLABEL_CSV_CANDIDATES = [\n    Path('/kaggle/input/datasets/benxelua/final-label/annotations_all_merged_other_1000nf.csv'),\n    Path('/kaggle/input/clean-dataset/annotations_all_merged_other_1000nf.csv'),\n    Path('/kaggle/input/lung-yolo-code/annotations_all_merged_other_1000nf.csv'),\n    Path('/kaggle/input/LUNG_YOLO_CODE/annotations_all_merged_other_1000nf.csv'),\n    Path('LUNG_YOLO_CODE/annotations_all_merged_other_1000nf.csv'),\n]\nLABEL_CSV = next((p for p in LABEL_CSV_CANDIDATES if p.exists()), LABEL_CSV_CANDIDATES[0])\n\n# output YOLO datasets, one subfolder per 3-channel preprocessing combo\nOUT_ROOT = Path(\"/kaggle/working/yolo_vindr_multiclass_combos\")\n\n# =========================\n# SUBSET CONFIG\n# =========================\nUSE_SMALL_SUBSET = False     # True = subset nhỏ, False = full dataset\nIMAGES_PER_CLASS = 2         # mỗi class lấy khoảng N ảnh trong small subset\nNO_FINDING_IMAGES = None     # None = giữ đúng toàn bộ No finding có trong LABEL_CSV 1000nf\nSEED = 42\n\n# Giống CSV runtime flow trong LUNG_YOLO_CODE/train.py\nSPLIT_RATIOS = {\n    \"train\": 0.70,\n    \"val\": 0.10,\n    \"test\": 0.20,\n}\n\n# =========================\n# IMAGE / PREPROCESS CONFIG\n# =========================\nIMG_SIZE = 512\nSAVE_AS_JPG = False          # yêu cầu hiện tại: export PNG 3-channel\nJPG_QUALITY = 95\nREBUILD_OUTPUT = True        # True = xóa dataset combo cũ trong OUT_ROOT trước khi export lại\nCLEANUP_AFTER_COMBO = True      # True = train/eval xong combo nào thì xóa images/labels combo đó để tiết kiệm disk\nENABLE_TRAIN_AUG = True\n\nLOW_PERCENTILE = 0.5\nHIGH_PERCENTILE = 99.5\n\nPREPROCESS_COMBOS = [\n    (\"raw_minmax\", \"voi_lut\", \"percentile\"),\n    (\"raw_minmax\", \"voi_lut\", \"clahe\"),\n    (\"raw_minmax\", \"percentile\", \"clahe\"),\n    (\"voi_lut\", \"percentile\", \"clahe\"),\n]\n\n# =========================\n# TRAIN CONFIG\n# =========================\nMODEL_NAME = \"yolov8n.pt\"\nTRAIN_EPOCHS = 50\nTRAIN_BATCH = 8\nEVAL_BATCH = 8","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:50:38.438046Z","iopub.status.busy":"2026-04-25T06:50:38.437511Z","iopub.status.idle":"2026-04-25T06:50:38.444804Z","shell.execute_reply":"2026-04-25T06:50:38.444017Z"},"papermill":{"duration":0.014642,"end_time":"2026-04-25T06:50:38.446231","exception":false,"start_time":"2026-04-25T06:50:38.431589","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"34f2c77d","cell_type":"markdown","source":"# Helper functions","metadata":{"papermill":{"duration":0.00451,"end_time":"2026-04-25T06:50:38.455232","exception":false,"start_time":"2026-04-25T06:50:38.450722","status":"completed"},"tags":[]}},{"id":"ea607b84","cell_type":"code","source":"def seed_everything(seed=42):\n    random.seed(seed)\n    np.random.seed(seed)\n\nseed_everything(SEED)\n\n\ndef find_dicom_files_for_image_ids(folder: Path, image_ids):\n    requested_ids = sorted({str(x).strip() for x in image_ids if pd.notna(x) and str(x).strip()})\n    matched = {}\n\n    for image_id in requested_ids:\n        direct_candidates = [\n            folder / f\"{image_id}.dicom\",\n            folder / f\"{image_id}.dcm\",\n            folder / image_id,\n        ]\n        for candidate in direct_candidates:\n            if candidate.is_file():\n                matched[image_id] = candidate\n                break\n\n    remaining = [image_id for image_id in requested_ids if image_id not in matched]\n    if remaining:\n        remaining_set = set(remaining)\n        for p in sorted(folder.rglob(\"*\")):\n            if not remaining_set:\n                break\n            if p.is_file() and p.suffix.lower() in {\".dicom\", \".dcm\"} and p.stem in remaining_set:\n                matched[p.stem] = p\n                remaining_set.remove(p.stem)\n\n    missing = [image_id for image_id in requested_ids if image_id not in matched]\n    matched_files = [matched[image_id] for image_id in requested_ids if image_id in matched]\n    return matched_files, matched, missing\n\n\nfrom pathlib import Path\nimport numpy as np\nimport pydicom\nfrom pydicom.multival import MultiValue\nfrom pydicom.pixel_data_handlers.util import apply_modality_lut, apply_voi_lut\n\n\ndef normalize_to_uint8(img):\n    img = np.asarray(img, dtype=np.float32)\n    img = np.nan_to_num(img, nan=0.0, posinf=0.0, neginf=0.0)\n    img_min = float(np.min(img))\n    img_max = float(np.max(img))\n    if img_max > img_min:\n        img = (img - img_min) / (img_max - img_min)\n    else:\n        img = np.zeros_like(img, dtype=np.float32)\n    return (img * 255.0).clip(0, 255).astype(np.uint8)\n\n\ndef fix_monochrome(img, ds):\n    if getattr(ds, \"PhotometricInterpretation\", \"\") == \"MONOCHROME1\":\n        img = np.max(img) - img\n    return img\n\n\ndef dicom_rescale_pixels(ds):\n    img = ds.pixel_array.astype(np.float32)\n    slope = float(getattr(ds, \"RescaleSlope\", 1.0))\n    intercept = float(getattr(ds, \"RescaleIntercept\", 0.0))\n    return img * slope + intercept\n\n\ndef first_float(value, default):\n    if value is None:\n        return float(default)\n    if hasattr(value, \"value\"):\n        value = value.value\n    if isinstance(value, MultiValue) or isinstance(value, (list, tuple, np.ndarray)):\n        if len(value) == 0:\n            return float(default)\n        value = value[0]\n    try:\n        return float(value)\n    except Exception:\n        return float(default)\n\n\ndef raw_minmax_from_ds(ds):\n    img = ds.pixel_array.astype(np.float32)\n    img = fix_monochrome(img, ds)\n    return normalize_to_uint8(img)\n\n\ndef voi_lut_from_ds(ds):\n    img = apply_modality_lut(ds.pixel_array, ds)\n    img = apply_voi_lut(img, ds).astype(np.float32)\n    img = fix_monochrome(img, ds)\n    return normalize_to_uint8(img)\n\n\ndef percentile_from_ds(ds, low=LOW_PERCENTILE, high=HIGH_PERCENTILE):\n    img = dicom_rescale_pixels(ds)\n    img = fix_monochrome(img, ds)\n\n    lo = float(np.percentile(img, low))\n    hi = float(np.percentile(img, high))\n    if hi <= lo:\n        hi = lo + 1.0\n\n    img = np.clip(img, lo, hi)\n    img = (img - lo) / (hi - lo)\n    return (img * 255.0).clip(0, 255).astype(np.uint8)\n\n\ndef clahe_from_ds(ds):\n    img = dicom_rescale_pixels(ds)\n    img = fix_monochrome(img, ds)\n\n    default_center = float(np.median(img))\n    default_width = float(np.std(img) * 4.0)\n    c_orig = first_float(ds.get(\"WindowCenter\", default_center), default_center)\n    w_orig = first_float(ds.get(\"WindowWidth\", default_width), default_width)\n\n    if w_orig < 1:\n        w_orig = float(np.std(img) * 4.0 + 1e-6)\n\n    img_min = c_orig - w_orig / 2.0\n    img_max = c_orig + w_orig / 2.0\n    if img_max <= img_min:\n        img_min = float(np.min(img))\n        img_max = float(np.max(img))\n        if img_max <= img_min:\n            return np.zeros_like(img, dtype=np.uint8)\n\n    img_windowed = np.clip(img, img_min, img_max)\n    img_8bit = ((img_windowed - img_min) / (img_max - img_min) * 255.0).clip(0, 255).astype(np.uint8)\n\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    return clahe.apply(img_8bit)\n\n\n# Các hàm path-level để có thể test riêng từng cách xử lý.\ndef read_dicom_raw_minmax(path: Path):\n    ds = pydicom.dcmread(str(path))\n    return raw_minmax_from_ds(ds)\n\n\ndef read_dicom_to_uint8(path: Path):\n    # Backward-compatible alias: raw min-max theo pixel gốc.\n    return read_dicom_raw_minmax(path)\n\n\ndef read_dicom_voi_lut(path: Path):\n    ds = pydicom.dcmread(str(path))\n    return voi_lut_from_ds(ds)\n\n\ndef read_dicom_percentile_clipping(path: Path, low=LOW_PERCENTILE, high=HIGH_PERCENTILE):\n    ds = pydicom.dcmread(str(path))\n    return percentile_from_ds(ds, low=low, high=high)\n\n\ndef read_dicom_with_clahe(path: Path):\n    ds = pydicom.dcmread(str(path))\n    img_clahe = clahe_from_ds(ds)\n    return cv2.cvtColor(img_clahe, cv2.COLOR_GRAY2BGR)\n\n\ndef read_dicom_preprocess_variants(path: Path):\n    ds = pydicom.dcmread(str(path))\n    return {\n        \"raw_minmax\": raw_minmax_from_ds(ds),\n        \"voi_lut\": voi_lut_from_ds(ds),\n        \"percentile\": percentile_from_ds(ds, low=LOW_PERCENTILE, high=HIGH_PERCENTILE),\n        \"clahe\": clahe_from_ds(ds),\n    }\n\n\ndef combo_to_name(combo):\n    return \"__\".join(combo)\n\n\ndef build_preprocess_combo_image(variants, combo):\n    channels = []\n    for method_name in combo:\n        channel = variants[method_name]\n        if channel.ndim == 3:\n            channel = cv2.cvtColor(channel, cv2.COLOR_BGR2GRAY)\n        channels.append(channel.astype(np.uint8))\n    return np.stack(channels, axis=-1)\n\n\ndef resize_image_keep_shape(img, size=1024):\n    h, w = img.shape[:2]\n    resized = cv2.resize(img, (size, size), interpolation=cv2.INTER_AREA)\n    return resized, w, h","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:50:38.46542Z","iopub.status.busy":"2026-04-25T06:50:38.465034Z","iopub.status.idle":"2026-04-25T06:50:38.476761Z","shell.execute_reply":"2026-04-25T06:50:38.476015Z"},"papermill":{"duration":0.018574,"end_time":"2026-04-25T06:50:38.478239","exception":false,"start_time":"2026-04-25T06:50:38.459665","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"d7a63b58","cell_type":"markdown","source":"# Đọc CSV và chuẩn hóa annotation","metadata":{"papermill":{"duration":0.004458,"end_time":"2026-04-25T06:50:38.487289","exception":false,"start_time":"2026-04-25T06:50:38.482831","status":"completed"},"tags":[]}},{"id":"0208e182","cell_type":"code","source":"# Label fusion helpers for VinDr-CXR training annotations\nfrom dataclasses import dataclass\n\n@dataclass(frozen=True)\nclass FusionConfig:\n    iou_threshold: float = 0.5\n    ios_threshold: float = 0.7\n    merge_mode: str = \"median\"  # options: median, mean, enclosing\n\n\ndef _box_area(box):\n    width = max(0.0, float(box[2]) - float(box[0]))\n    height = max(0.0, float(box[3]) - float(box[1]))\n    return width * height\n\n\ndef _intersection_area(box_a, box_b):\n    x1 = max(float(box_a[0]), float(box_b[0]))\n    y1 = max(float(box_a[1]), float(box_b[1]))\n    x2 = min(float(box_a[2]), float(box_b[2]))\n    y2 = min(float(box_a[3]), float(box_b[3]))\n    width = max(0.0, x2 - x1)\n    height = max(0.0, y2 - y1)\n    return width * height\n\n\ndef compute_iou(box_a, box_b):\n    inter = _intersection_area(box_a, box_b)\n    if inter <= 0.0:\n        return 0.0\n    union = _box_area(box_a) + _box_area(box_b) - inter\n    if union <= 0.0:\n        return 0.0\n    return inter / union\n\n\ndef compute_ios(box_a, box_b):\n    inter = _intersection_area(box_a, box_b)\n    if inter <= 0.0:\n        return 0.0\n    smaller = min(_box_area(box_a), _box_area(box_b))\n    if smaller <= 0.0:\n        return 0.0\n    return inter / smaller\n\n\ndef should_link_boxes(box_a, box_b, config):\n    return (\n        compute_iou(box_a, box_b) >= config.iou_threshold\n        or compute_ios(box_a, box_b) >= config.ios_threshold\n    )\n\n\ndef merge_cluster_boxes(boxes, merge_mode=\"median\"):\n    boxes = np.asarray(boxes, dtype=float)\n    if len(boxes) == 1:\n        return boxes[0].astype(float)\n\n    if merge_mode == \"median\":\n        merged = np.median(boxes, axis=0)\n    elif merge_mode == \"mean\":\n        merged = np.mean(boxes, axis=0)\n    elif merge_mode == \"enclosing\":\n        merged = np.array([\n            np.min(boxes[:, 0]),\n            np.min(boxes[:, 1]),\n            np.max(boxes[:, 2]),\n            np.max(boxes[:, 3]),\n        ], dtype=float)\n    else:\n        raise ValueError(f\"Unsupported merge_mode: {merge_mode}\")\n\n    if merged[2] <= merged[0]:\n        merged[2] = merged[0] + 1e-6\n    if merged[3] <= merged[1]:\n        merged[3] = merged[1] + 1e-6\n    return merged.astype(float)\n\n\ndef _connected_components(indices, adjacency):\n    components = []\n    visited = set()\n\n    for start in indices:\n        if start in visited:\n            continue\n        stack = [start]\n        component = []\n        visited.add(start)\n\n        while stack:\n            node = stack.pop()\n            component.append(node)\n            for neighbor in adjacency[node]:\n                if neighbor not in visited:\n                    visited.add(neighbor)\n                    stack.append(neighbor)\n\n        components.append(sorted(component))\n\n    return components\n\n\ndef fuse_same_class_boxes(class_df, bbox_cols=(\"x_min\", \"y_min\", \"x_max\", \"y_max\"), config=None):\n    config = config or FusionConfig()\n    if len(class_df) <= 1:\n        out = class_df.copy()\n        out[\"fusion_group_size\"] = 1\n        out[\"was_fused\"] = False\n        return out\n\n    coords = class_df.loc[:, list(bbox_cols)].to_numpy(dtype=float)\n    indices = list(range(len(class_df)))\n    adjacency = {idx: set() for idx in indices}\n\n    for i in range(len(coords)):\n        for j in range(i + 1, len(coords)):\n            if should_link_boxes(coords[i], coords[j], config):\n                adjacency[i].add(j)\n                adjacency[j].add(i)\n\n    components = _connected_components(indices, adjacency)\n    fused_rows = []\n\n    for component in components:\n        component_df = class_df.iloc[component].copy()\n        merged_box = merge_cluster_boxes(\n            component_df.loc[:, list(bbox_cols)].to_numpy(dtype=float),\n            merge_mode=config.merge_mode,\n        )\n\n        row = component_df.iloc[0].copy()\n        row[bbox_cols[0]] = float(merged_box[0])\n        row[bbox_cols[1]] = float(merged_box[1])\n        row[bbox_cols[2]] = float(merged_box[2])\n        row[bbox_cols[3]] = float(merged_box[3])\n        row[\"fusion_group_size\"] = int(len(component))\n        row[\"was_fused\"] = bool(len(component) > 1)\n        fused_rows.append(row)\n\n    return pd.DataFrame(fused_rows).reset_index(drop=True)\n\n\ndef fuse_annotation_dataframe(\n    df,\n    image_col=\"image_id\",\n    class_col=\"class_name\",\n    bbox_cols=(\"x_min\", \"y_min\", \"x_max\", \"y_max\"),\n    config=None,\n):\n    config = config or FusionConfig()\n    fused_groups = []\n    grouped = df.groupby([image_col, class_col], sort=False, group_keys=False)\n    for _, group_df in grouped:\n        fused_groups.append(\n            fuse_same_class_boxes(\n                class_df=group_df,\n                bbox_cols=bbox_cols,\n                config=config,\n            )\n        )\n\n    if not fused_groups:\n        out = df.copy()\n        out[\"fusion_group_size\"] = 1\n        out[\"was_fused\"] = False\n        return out\n\n    return pd.concat(fused_groups, ignore_index=True).reset_index(drop=True)\n\n","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:50:38.497514Z","iopub.status.busy":"2026-04-25T06:50:38.497132Z","iopub.status.idle":"2026-04-25T06:50:38.515925Z","shell.execute_reply":"2026-04-25T06:50:38.515309Z"},"papermill":{"duration":0.025669,"end_time":"2026-04-25T06:50:38.517308","exception":false,"start_time":"2026-04-25T06:50:38.491639","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"48a3d052","cell_type":"code","source":"# Label handling ported from LUNG_YOLO_CODE/train.py CSV runtime flow.\nCSV_BACKGROUND_CLASS = \"No finding\"\nCSV_BBOX_FORMAT = \"auto\"   # auto, xyxy_pixel, xyxy_norm\nCSV_SCORE_COL = \"score\"\nCSV_IMAGE_WIDTH_COL = \"\"\nCSV_IMAGE_HEIGHT_COL = \"\"\nWBF_IOU_THRESHOLD = 0.5\nWBF_SKIP_BOX_THRESHOLD = 0.0\n\n\ndef is_background_class(class_name):\n    normalized = str(class_name).strip().lower().replace(\"_\", \" \")\n    background_names = {\n        CSV_BACKGROUND_CLASS.strip().lower().replace(\"_\", \" \"),\n        \"no finding\",\n        \"background\",\n        \"negative\",\n    }\n    return normalized in background_names\n\n\ndef first_existing_column(df, explicit_col, candidates):\n    if explicit_col and explicit_col in df.columns:\n        return explicit_col\n    for col in candidates:\n        if col in df.columns:\n            return col\n    return None\n\n\ndef normalize_label_csv(csv_path: Path):\n    csv_path = Path(csv_path)\n    if not csv_path.exists():\n        candidate_text = \"\\n\".join(f\"- {p}\" for p in LABEL_CSV_CANDIDATES)\n        raise FileNotFoundError(\n            f\"Không thấy LABEL_CSV merged 1000nf: {csv_path}\\n\"\n            f\"Các path đã thử:\\n{candidate_text}\"\n        )\n\n    out = pd.read_csv(csv_path)\n    print(\"CSV label path:\", csv_path)\n    print(\"CSV shape:\", out.shape)\n    print(\"CSV columns:\", out.columns.tolist())\n    display(out.head())\n\n    required_cols = [\"image_id\", \"class_name\", \"x_min\", \"y_min\", \"x_max\", \"y_max\"]\n    missing_cols = [c for c in required_cols if c not in out.columns]\n    if missing_cols:\n        raise ValueError(f\"CSV thiếu cột bắt buộc: {missing_cols}\")\n\n    out = out.copy()\n    out[\"image_id\"] = out[\"image_id\"].astype(\"string\").str.strip()\n    out[\"class_name\"] = out[\"class_name\"].astype(\"string\").str.strip()\n    out = out.dropna(subset=[\"image_id\", \"class_name\"]).copy()\n\n    source_col = first_existing_column(out, \"source\", [\"source\", \"original_split\", \"split\"])\n    if source_col:\n        out[\"source\"] = out[source_col].astype(\"string\").str.strip().str.lower()\n        print(\"CSV source column:\", source_col)\n    else:\n        out[\"source\"] = \"train\"\n        print(\"CSV source column: not found; fallback source=train\")\n\n    # DICOM competition chỉ có ROOT/train và ROOT/test. Nếu CSV có val/validation thì đọc từ ROOT/test.\n    out[\"source\"] = out[\"source\"].replace({\"validation\": \"test\", \"valid\": \"test\", \"val\": \"test\"})\n\n    sample_id_col = first_existing_column(out, \"sample_id\", [\"sample_id\"])\n    if sample_id_col:\n        base_sample_id = out[sample_id_col].astype(\"string\").str.strip().astype(str)\n    else:\n        base_sample_id = out[\"image_id\"].astype(str)\n\n    # Notebook dùng source__id để tránh trùng tên khi đọc DICOM từ 2 folder gốc.\n    if base_sample_id.str.contains(\"__\", regex=False).any():\n        out[\"sample_id\"] = base_sample_id\n    else:\n        out[\"sample_id\"] = out[\"source\"].astype(str) + \"__\" + base_sample_id\n\n    filename_col = first_existing_column(out, \"\", [\"file_name\", \"filename\", \"image_path\", \"path\"])\n    out[\"csv_file_name\"] = out[filename_col].astype(\"string\").str.strip() if filename_col else \"\"\n\n    width_col = first_existing_column(out, CSV_IMAGE_WIDTH_COL, [\"image_width\", \"img_width\", \"original_width\", \"orig_width\", \"width\"])\n    height_col = first_existing_column(out, CSV_IMAGE_HEIGHT_COL, [\"image_height\", \"img_height\", \"original_height\", \"orig_height\", \"height\"])\n    if width_col and height_col:\n        out[\"source_width\"] = pd.to_numeric(out[width_col], errors=\"coerce\")\n        out[\"source_height\"] = pd.to_numeric(out[height_col], errors=\"coerce\")\n        print(\"CSV image size columns:\", width_col, height_col)\n    else:\n        out[\"source_width\"] = np.nan\n        out[\"source_height\"] = np.nan\n        print(\"CSV image size columns: not found; pixel boxes will use DICOM size\")\n\n    if CSV_SCORE_COL and CSV_SCORE_COL in out.columns:\n        out[\"box_score\"] = pd.to_numeric(out[CSV_SCORE_COL], errors=\"coerce\").fillna(1.0)\n        print(\"CSV WBF score column:\", CSV_SCORE_COL)\n    else:\n        out[\"box_score\"] = 1.0\n        print(\"CSV WBF score column: not found; all boxes weight=1\")\n\n    for c in [\"x_min\", \"y_min\", \"x_max\", \"y_max\"]:\n        out[c] = pd.to_numeric(out[c], errors=\"coerce\")\n\n    out[\"is_background\"] = out[\"class_name\"].map(is_background_class)\n    return out.reset_index(drop=True)\n\n\ndef compute_iou_xyxy(box_a, box_b):\n    x1 = max(float(box_a[0]), float(box_b[0]))\n    y1 = max(float(box_a[1]), float(box_b[1]))\n    x2 = min(float(box_a[2]), float(box_b[2]))\n    y2 = min(float(box_a[3]), float(box_b[3]))\n    inter_w = max(0.0, x2 - x1)\n    inter_h = max(0.0, y2 - y1)\n    inter = inter_w * inter_h\n    if inter <= 0.0:\n        return 0.0\n    area_a = max(0.0, float(box_a[2]) - float(box_a[0])) * max(0.0, float(box_a[3]) - float(box_a[1]))\n    area_b = max(0.0, float(box_b[2]) - float(box_b[0])) * max(0.0, float(box_b[3]) - float(box_b[1]))\n    union = area_a + area_b - inter\n    if union <= 0.0:\n        return 0.0\n    return inter / union\n\n\ndef weighted_fuse_group(group_df):\n    if len(group_df) <= 1:\n        out = group_df.copy()\n        out[\"fusion_group_size\"] = 1\n        out[\"was_fused\"] = False\n        return out\n\n    rows = []\n    work = group_df.copy()\n    work[\"_score_sort\"] = pd.to_numeric(work[\"box_score\"], errors=\"coerce\").fillna(1.0)\n    work = work.sort_values(\"_score_sort\", ascending=False).reset_index(drop=True)\n\n    clusters = []\n    for idx, row in work.iterrows():\n        box = row[[\"x_min\", \"y_min\", \"x_max\", \"y_max\"]].to_numpy(dtype=float)\n        score = float(row.get(\"box_score\", 1.0))\n        best_cluster = None\n        best_iou = 0.0\n\n        for cluster_idx, cluster in enumerate(clusters):\n            iou = compute_iou_xyxy(box, cluster[\"box\"])\n            if iou >= WBF_IOU_THRESHOLD and iou > best_iou:\n                best_iou = iou\n                best_cluster = cluster_idx\n\n        if best_cluster is None:\n            clusters.append({\n                \"indices\": [idx],\n                \"boxes\": [box],\n                \"scores\": [score],\n                \"box\": box.copy(),\n            })\n            continue\n\n        cluster = clusters[best_cluster]\n        cluster[\"indices\"].append(idx)\n        cluster[\"boxes\"].append(box)\n        cluster[\"scores\"].append(score)\n        weights = np.asarray(cluster[\"scores\"], dtype=float)\n        weights = np.maximum(weights, 1e-6)\n        cluster[\"box\"] = np.average(np.asarray(cluster[\"boxes\"], dtype=float), axis=0, weights=weights)\n\n    for cluster in clusters:\n        fused = work.loc[cluster[\"indices\"]].iloc[0].copy()\n        weights = np.asarray(cluster[\"scores\"], dtype=float)\n        weights = np.maximum(weights, 1e-6)\n        fused_box = np.average(np.asarray(cluster[\"boxes\"], dtype=float), axis=0, weights=weights)\n        fused[\"x_min\"] = float(fused_box[0])\n        fused[\"y_min\"] = float(fused_box[1])\n        fused[\"x_max\"] = float(fused_box[2])\n        fused[\"y_max\"] = float(fused_box[3])\n        fused[\"box_score\"] = float(np.mean(cluster[\"scores\"]))\n        fused[\"fusion_group_size\"] = int(len(cluster[\"indices\"]))\n        fused[\"was_fused\"] = bool(len(cluster[\"indices\"]) > 1)\n        rows.append(fused.drop(labels=[\"_score_sort\"], errors=\"ignore\"))\n\n    return pd.DataFrame(rows)\n\n\ndef apply_wbf_to_annotations(ann_df):\n    fused_groups = []\n    grouped = ann_df.groupby([\"sample_id\", \"class_name\"], sort=False, group_keys=False)\n    for _, group in grouped:\n        fused_groups.append(weighted_fuse_group(group))\n\n    if not fused_groups:\n        return ann_df.copy()\n\n    fused_df = pd.concat(fused_groups, ignore_index=True).reset_index(drop=True)\n    print(\"WBF enabled\")\n    print(\"WBF IoU threshold:\", WBF_IOU_THRESHOLD)\n    print(\"Boxes before WBF:\", len(ann_df))\n    print(\"Boxes after WBF:\", len(fused_df))\n    print(\"Rows created from fused clusters:\", int(fused_df.get(\"was_fused\", pd.Series(dtype=bool)).sum()))\n    return fused_df\n\n\ndf_csv_all = normalize_label_csv(LABEL_CSV)\n\nall_sample_ids = set(df_csv_all[\"sample_id\"].dropna().astype(str).str.strip().unique().tolist())\nall_csv_ids = all_sample_ids\nno_finding_sample_ids = set(\n    df_csv_all.loc[df_csv_all[\"is_background\"], \"sample_id\"].dropna().astype(str).tolist()\n)\n\nprint(\"Merged 1000nf CSV shape:\", df_csv_all.shape)\nprint(\"Merged 1000nf unique sample_id:\", len(all_sample_ids))\nprint(\"No finding/background sample_id in LABEL_CSV:\", len(no_finding_sample_ids))\nprint(\"Source distribution from LABEL_CSV:\")\ndisplay(df_csv_all.drop_duplicates(\"sample_id\")[\"source\"].value_counts().rename_axis(\"source\").reset_index(name=\"images\"))\n\nann_df = df_csv_all[~df_csv_all[\"is_background\"]].copy()\nann_df = ann_df.dropna(subset=[\"image_id\", \"sample_id\", \"class_name\", \"x_min\", \"y_min\", \"x_max\", \"y_max\"]).copy()\nann_df = ann_df[\n    (ann_df[\"x_max\"] > ann_df[\"x_min\"])\n    & (ann_df[\"y_max\"] > ann_df[\"y_min\"])\n    & (ann_df[\"box_score\"] >= WBF_SKIP_BOX_THRESHOLD)\n].copy()\n\ndf_before_fusion = ann_df.copy()\ndf_full = apply_wbf_to_annotations(ann_df)\n\nCLASS_NAMES = sorted(df_full[\"class_name\"].astype(str).unique().tolist())\nif not CLASS_NAMES:\n    raise ValueError(\"Không có class bất thường hợp lệ sau khi bỏ No finding/background\")\nCLASS2ID = {name: i for i, name in enumerate(CLASS_NAMES)}\nID2CLASS = {i: name for name, i in CLASS2ID.items()}\n\nabnormal_sample_ids = set(df_full[\"sample_id\"].astype(str).unique())\nbackground_sample_ids = sorted(all_sample_ids - abnormal_sample_ids)\nno_finding_only_sample_ids = sorted(set(no_finding_sample_ids) & set(background_sample_ids))\n\nprint(\"Num classes:\", len(CLASS_NAMES))\nprint(\"CLASS_NAMES:\", CLASS_NAMES)\nprint(\"CLASS2ID:\", CLASS2ID)\nprint(\"Total unique sample_id in LABEL_CSV:\", len(all_sample_ids))\nprint(\"Images with >=1 valid abnormal bbox:\", len(abnormal_sample_ids))\nprint(\"Background-only sample_id:\", len(background_sample_ids))\nprint(\"No finding only sample_id:\", len(no_finding_only_sample_ids))\ndisplay(\n    df_full.loc[df_full.get(\"was_fused\", False), [\"source\", \"image_id\", \"sample_id\", \"class_name\", \"fusion_group_size\"]]\n    .head(10)\n    .reset_index(drop=True)\n)","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:50:38.527332Z","iopub.status.busy":"2026-04-25T06:50:38.527124Z","iopub.status.idle":"2026-04-25T06:51:23.126282Z","shell.execute_reply":"2026-04-25T06:51:23.125575Z"},"papermill":{"duration":44.605983,"end_time":"2026-04-25T06:51:23.127791","exception":false,"start_time":"2026-04-25T06:50:38.521808","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"9918bec6","cell_type":"code","source":"# # Debug visualize label fusion before vs after for one training image\n# import matplotlib.pyplot as plt\n# from matplotlib.patches import Rectangle\n\n# DEBUG_FUSION_IMAGE_ID = \"\"  # ví dụ: \"0005e8e3701dfb1dd93d53e2ff537b6e\"\n# SHOW_CLASS_TEXT = True\n# FIGSIZE = (18, 9)\n\n# if not DEBUG_FUSION_IMAGE_ID:\n#     raise ValueError(\"Điền DEBUG_FUSION_IMAGE_ID trước khi chạy cell này\")\n# if 'df_before_fusion' not in globals():\n#     raise ValueError(\"Chạy cell fusion trước để tạo df_before_fusion và df_full\")\n\n\n# def find_train_dicom_path(train_dir: Path, image_id: str):\n#     direct_candidates = [\n#         train_dir / f\"{image_id}.dicom\",\n#         train_dir / f\"{image_id}.dcm\",\n#         train_dir / image_id,\n#     ]\n#     for candidate in direct_candidates:\n#         if candidate.is_file():\n#             return candidate\n\n#     for p in sorted(train_dir.rglob(\"*\")):\n#         if p.is_file() and p.suffix.lower() in {\".dicom\", \".dcm\"} and p.stem == image_id:\n#             return p\n#     return None\n\n\n# def draw_boxes(ax, boxes_df, color, title_prefix=\"\"):\n#     for _, row in boxes_df.iterrows():\n#         x1, y1, x2, y2 = float(row['x_min']), float(row['y_min']), float(row['x_max']), float(row['y_max'])\n#         rect = Rectangle((x1, y1), x2 - x1, y2 - y1, fill=False, edgecolor=color, linewidth=2)\n#         ax.add_patch(rect)\n#         if SHOW_CLASS_TEXT:\n#             label = str(row['class_name'])\n#             if 'fusion_group_size' in row and pd.notna(row.get('fusion_group_size', np.nan)):\n#                 if int(row.get('fusion_group_size', 1)) > 1:\n#                     label = f\"{label} (n={int(row['fusion_group_size'])})\"\n#             ax.text(\n#                 x1,\n#                 max(5.0, y1 - 4.0),\n#                 f\"{title_prefix}{label}\",\n#                 color=color,\n#                 fontsize=9,\n#                 bbox=dict(facecolor='black', alpha=0.5, pad=1),\n#             )\n\n\n# before_df = df_before_fusion[df_before_fusion['image_id'].astype(str) == str(DEBUG_FUSION_IMAGE_ID)].copy()\n# after_df = df_full[df_full['image_id'].astype(str) == str(DEBUG_FUSION_IMAGE_ID)].copy()\n\n# print(f\"image_id = {DEBUG_FUSION_IMAGE_ID}\")\n# print(f\"Boxes before fusion: {len(before_df)}\")\n# print(f\"Boxes after fusion: {len(after_df)}\")\n\n# if len(before_df) == 0 and len(after_df) == 0:\n#     raise ValueError(f\"Không tìm thấy annotation cho image_id={DEBUG_FUSION_IMAGE_ID}\")\n\n# display(before_df[['image_id', 'class_name', 'x_min', 'y_min', 'x_max', 'y_max']].reset_index(drop=True))\n# display(after_df[['image_id', 'class_name', 'x_min', 'y_min', 'x_max', 'y_max', 'fusion_group_size', 'was_fused']].reset_index(drop=True))\n\n# dicom_path = find_train_dicom_path(TRAIN_DIR, str(DEBUG_FUSION_IMAGE_ID))\n# if dicom_path is None:\n#     raise FileNotFoundError(f\"Không tìm thấy file DICOM cho image_id={DEBUG_FUSION_IMAGE_ID} trong {TRAIN_DIR}\")\n\n# img = read_dicom_to_uint8(dicom_path)\n\n# fig, axes = plt.subplots(1, 2, figsize=FIGSIZE)\n# for ax in axes:\n#     if img.ndim == 2:\n#         ax.imshow(img, cmap='gray')\n#     else:\n#         ax.imshow(img)\n#     ax.axis('off')\n\n# axes[0].set_title(f\"Before fusion: {DEBUG_FUSION_IMAGE_ID}\")\n# draw_boxes(axes[0], before_df, color='red', title_prefix='before: ')\n\n# axes[1].set_title(f\"After fusion: {DEBUG_FUSION_IMAGE_ID}\")\n# draw_boxes(axes[1], after_df, color='lime', title_prefix='after: ')\n\n# plt.tight_layout()\n# plt.show()\n\n","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:51:23.140319Z","iopub.status.busy":"2026-04-25T06:51:23.139905Z","iopub.status.idle":"2026-04-25T06:51:23.144595Z","shell.execute_reply":"2026-04-25T06:51:23.144035Z"},"papermill":{"duration":0.012452,"end_time":"2026-04-25T06:51:23.146002","exception":false,"start_time":"2026-04-25T06:51:23.13355","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"61b21fe5","cell_type":"markdown","source":"# Chọn subset nhỏ theo từng class từ train + test đã gộp","metadata":{"papermill":{"duration":0.005189,"end_time":"2026-04-25T06:51:23.156252","exception":false,"start_time":"2026-04-25T06:51:23.151063","status":"completed"},"tags":[]}},{"id":"f5f6c192","cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nprint(\"Building subset from merged train + test CSV...\")\n\n# subset theo annotation bbox hợp lệ, nhưng full mode vẫn lấy toàn bộ sample_id từ CSV.\ndf_work = df_full.copy()\n\nclass_to_ids = (\n    df_work.groupby(\"class_name\")[\"sample_id\"]\n    .unique()\n    .to_dict()\n)\n\n\ndef build_small_subset_by_class_fast(class_to_ids, images_per_class=12, seed=42, background_ids=None):\n    rng = np.random.default_rng(seed)\n\n    selected_ids = set()\n    per_class_summary = []\n\n    for cls in sorted(class_to_ids.keys()):\n        cls_ids = list(class_to_ids[cls])\n        n_take = min(images_per_class, len(cls_ids))\n\n        if n_take > 0:\n            chosen = rng.choice(cls_ids, size=n_take, replace=False).tolist()\n        else:\n            chosen = []\n\n        selected_ids.update(chosen)\n\n        per_class_summary.append({\n            \"class_name\": cls,\n            \"available_images\": len(cls_ids),\n            \"selected_images\": len(chosen)\n        })\n\n    background_ids = sorted(set(background_ids or []))\n    n_take_bg = min(images_per_class, len(background_ids))\n    if n_take_bg > 0:\n        chosen_bg = rng.choice(background_ids, size=n_take_bg, replace=False).tolist()\n    else:\n        chosen_bg = []\n    selected_ids.update(chosen_bg)\n    per_class_summary.append({\n        \"class_name\": \"No finding\",\n        \"available_images\": len(background_ids),\n        \"selected_images\": len(chosen_bg),\n    })\n\n    return selected_ids, pd.DataFrame(per_class_summary)\n\n\nrng_subset = np.random.default_rng(SEED)\nbackground_pool = sorted(set(background_sample_ids))\nif NO_FINDING_IMAGES is None:\n    selected_background_ids = set(background_pool)\nelse:\n    n_no_finding_take = min(int(NO_FINDING_IMAGES), len(background_pool))\n    selected_background_ids = set(\n        rng_subset.choice(background_pool, size=n_no_finding_take, replace=False).tolist()\n    ) if n_no_finding_take > 0 else set()\n\nif USE_SMALL_SUBSET:\n    chosen_ids, subset_summary_df = build_small_subset_by_class_fast(\n        class_to_ids,\n        images_per_class=IMAGES_PER_CLASS,\n        seed=SEED,\n        background_ids=background_pool,\n    )\n    chosen_ids = set(map(str, chosen_ids)) & all_sample_ids\n    print(\"Using SMALL subset mode\")\n    print(\"Total selected unique samples from merged CSV:\", len(chosen_ids))\n    display(subset_summary_df)\nelse:\n    # Full mode: dùng đúng LABEL_CSV 1000nf giống train.py.\n    # Nếu NO_FINDING_IMAGES=None thì giữ toàn bộ background đã có sẵn trong LABEL_CSV.\n    chosen_ids = set(abnormal_sample_ids) | selected_background_ids\n    print(\"Using FULL dataset mode from LABEL_CSV 1000nf\")\n    print(\"Abnormal samples selected:\", len(abnormal_sample_ids))\n    print(\"Background/No finding samples selected:\", len(selected_background_ids), \"of\", len(background_pool))\n    print(\"NO_FINDING_IMAGES:\", NO_FINDING_IMAGES)\n    print(\"Total selected unique samples from LABEL_CSV:\", len(chosen_ids))\n\nprint(\"Matching requested DICOM files from both source folders...\")\n\nSOURCE_DIRS = {\n    \"train\": TRAIN_DIR,\n    \"test\": TEST_DIR,\n}\n\nsample_lookup_df = (\n    df_csv_all[[\"sample_id\", \"source\", \"image_id\"]]\n    .drop_duplicates(\"sample_id\")\n    .reset_index(drop=True)\n)\n\n\ndef find_dicom_records_for_sample_ids(sample_ids, sample_lookup_df, source_dirs):\n    requested = set(map(str, sample_ids))\n    lookup = sample_lookup_df[sample_lookup_df[\"sample_id\"].isin(requested)].copy()\n\n    records = []\n    path_map = {}\n    missing_sample_ids = []\n\n    for source, group in lookup.groupby(\"source\", sort=False):\n        source = str(source).strip().lower()\n        if source not in source_dirs:\n            print(f\"Unknown source={source}; mark samples as missing\")\n            missing_sample_ids.extend(group[\"sample_id\"].astype(str).tolist())\n            continue\n\n        source_dir = Path(source_dirs[source])\n        if not source_dir.exists():\n            missing_sample_ids.extend(group[\"sample_id\"].astype(str).tolist())\n            continue\n\n        image_ids = group[\"image_id\"].astype(str).tolist()\n        _, image_path_map, missing_image_ids = find_dicom_files_for_image_ids(source_dir, image_ids)\n        missing_image_ids = set(missing_image_ids)\n\n        for row in group.itertuples(index=False):\n            sample_id = str(row.sample_id)\n            image_id = str(row.image_id)\n            if image_id in image_path_map:\n                record = {\n                    \"sample_id\": sample_id,\n                    \"source\": str(row.source),\n                    \"image_id\": image_id,\n                    \"path\": image_path_map[image_id],\n                }\n                records.append(record)\n                path_map[sample_id] = image_path_map[image_id]\n            elif image_id in missing_image_ids:\n                missing_sample_ids.append(sample_id)\n\n    records = sorted(records, key=lambda x: x[\"sample_id\"])\n    missing_sample_ids = sorted(set(missing_sample_ids) | (requested - set(path_map.keys())))\n    return records, path_map, missing_sample_ids\n\n\ndicom_records_selected, dicom_path_map, missing_dicom_ids = find_dicom_records_for_sample_ids(\n    chosen_ids,\n    sample_lookup_df=sample_lookup_df,\n    source_dirs=SOURCE_DIRS,\n)\ndicom_files_selected = [record[\"path\"] for record in dicom_records_selected]\nused_sample_ids = sorted(dicom_path_map.keys())\nused_image_ids = used_sample_ids  # alias cũ, giờ là sample_id\n\nprint(\"Requested merged sample_ids:\", len(chosen_ids))\nprint(\"Matched DICOM records:\", len(dicom_records_selected))\nprint(\"Missing requested DICOM samples:\", len(missing_dicom_ids))\nif missing_dicom_ids:\n    print(\"First missing sample_ids:\", missing_dicom_ids[:10])\n\nsource_summary = pd.DataFrame(dicom_records_selected)[\"source\"].value_counts().rename_axis(\"source\").reset_index(name=\"matched_images\")\ndisplay(source_summary)","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:51:23.167818Z","iopub.status.busy":"2026-04-25T06:51:23.16747Z","iopub.status.idle":"2026-04-25T06:51:44.495451Z","shell.execute_reply":"2026-04-25T06:51:44.494599Z"},"papermill":{"duration":21.33568,"end_time":"2026-04-25T06:51:44.496972","exception":false,"start_time":"2026-04-25T06:51:23.161292","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"24d9f3b6","cell_type":"code","source":"used_sample_ids = sorted(dicom_path_map.keys())\nused_image_ids = used_sample_ids  # alias cũ, giờ là sample_id\n\ndf = df_full[df_full[\"sample_id\"].astype(str).str.strip().isin(used_sample_ids)].copy()\nused_negative_only_ids = sorted(set(used_sample_ids) - set(df[\"sample_id\"].astype(str).unique()))\n\nprint(\"Requested merged sample_ids:\", len(chosen_ids))\nprint(\"Used DICOM records:\", len(dicom_records_selected))\nprint(\"Selected annotation rows:\", len(df))\nprint(\"Used annotated sample_id:\", df[\"sample_id\"].nunique())\nprint(\"Used negative-only sample_id:\", len(used_negative_only_ids))\nprint(\"Missing requested sample_ids:\", len(missing_dicom_ids))","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:51:44.509644Z","iopub.status.busy":"2026-04-25T06:51:44.509405Z","iopub.status.idle":"2026-04-25T06:51:44.53469Z","shell.execute_reply":"2026-04-25T06:51:44.533928Z"},"papermill":{"duration":0.033367,"end_time":"2026-04-25T06:51:44.536168","exception":false,"start_time":"2026-04-25T06:51:44.502801","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"4f3fc3aa","cell_type":"markdown","source":"# Stratified split train/val/test","metadata":{"papermill":{"duration":0.005641,"end_time":"2026-04-25T06:51:44.547198","exception":false,"start_time":"2026-04-25T06:51:44.541557","status":"completed"},"tags":[]}},{"id":"cf094048","cell_type":"code","source":"from collections import Counter, defaultdict\nimport numpy as np\nimport pandas as pd\n\nBACKGROUND_CLASS = \"No finding\"\n\n\ndef normalize_split_ratios(split_ratios):\n    split_ratios = dict(split_ratios)\n    total = sum(float(v) for v in split_ratios.values())\n    if total <= 0:\n        raise ValueError(\"SPLIT_RATIOS phải có tổng > 0\")\n    return {k: float(v) / total for k, v in split_ratios.items()}\n\n\ndef compute_target_sizes(n_items, split_ratios):\n    raw = {split: n_items * ratio for split, ratio in split_ratios.items()}\n    sizes = {split: int(np.floor(value)) for split, value in raw.items()}\n    remaining = n_items - sum(sizes.values())\n\n    order = sorted(split_ratios.keys(), key=lambda split: (raw[split] - sizes[split], split), reverse=True)\n    for split in order[:remaining]:\n        sizes[split] += 1\n\n    nonzero_splits = [split for split, ratio in split_ratios.items() if ratio > 0]\n    if n_items >= len(nonzero_splits):\n        for split in nonzero_splits:\n            if sizes[split] == 0:\n                donor = max(sizes, key=sizes.get)\n                sizes[donor] -= 1\n                sizes[split] += 1\n\n    return sizes\n\n\ndef build_image_class_sets(df, image_ids, image_col=\"sample_id\", class_col=\"class_name\"):\n    image_ids = sorted(set(map(str, image_ids)))\n    grouped = (\n        df.groupby(image_col)[class_col]\n        .apply(lambda x: sorted(set(map(str, x))))\n        .to_dict()\n    )\n    return {\n        image_id: grouped.get(image_id, [BACKGROUND_CLASS])\n        for image_id in image_ids\n    }\n\n\ndef build_multilabel_stratified_split(df, image_ids, split_ratios=None, seed=42):\n    split_ratios = normalize_split_ratios(split_ratios or {\"train\": 0.7, \"val\": 0.15, \"test\": 0.15})\n    rng = np.random.default_rng(seed)\n\n    image_ids = sorted(set(map(str, image_ids)))\n    image_class_sets = build_image_class_sets(df, image_ids)\n    split_names = list(split_ratios.keys())\n\n    target_sizes = compute_target_sizes(len(image_ids), split_ratios)\n    class_total = Counter(cls for classes in image_class_sets.values() for cls in classes)\n    target_class_counts = {\n        split: {cls: class_total[cls] * split_ratios[split] for cls in class_total}\n        for split in split_names\n    }\n\n    current_sizes = {split: 0 for split in split_names}\n    current_class_counts = {split: Counter() for split in split_names}\n    split_sets = {split: set() for split in split_names}\n\n    # Ưu tiên gán ảnh chứa class hiếm/multilabel trước để giữ phân bố class tốt hơn.\n    ordered_ids = sorted(\n        image_ids,\n        key=lambda image_id: (\n            -sum(1.0 / max(class_total[cls], 1) for cls in image_class_sets[image_id]),\n            -len(image_class_sets[image_id]),\n            rng.random(),\n        ),\n    )\n\n    for image_id in ordered_ids:\n        classes = image_class_sets[image_id]\n        candidates = [split for split in split_names if current_sizes[split] < target_sizes[split]]\n        if not candidates:\n            candidates = split_names\n\n        best_split = None\n        best_score = None\n\n        for split in candidates:\n            size_deficit = (target_sizes[split] - current_sizes[split]) / max(target_sizes[split], 1)\n            class_deficit = 0.0\n            over_penalty = 0.0\n\n            for cls in classes:\n                target = max(target_class_counts[split][cls], 1e-9)\n                before = current_class_counts[split][cls]\n                after = before + 1\n                class_deficit += max(target - before, 0.0) / target\n                over_penalty += max(after - target, 0.0) / target\n\n            score = size_deficit + class_deficit - over_penalty\n            tie_breaker = rng.random() * 1e-6\n            score += tie_breaker\n\n            if best_score is None or score > best_score:\n                best_score = score\n                best_split = split\n\n        split_sets[best_split].add(image_id)\n        current_sizes[best_split] += 1\n        for cls in classes:\n            current_class_counts[best_split][cls] += 1\n\n    return split_sets, image_class_sets\n\n\ndef summarize_split_distribution(split_sets, image_class_sets):\n    class_names = sorted({cls for classes in image_class_sets.values() for cls in classes})\n    rows = []\n    for cls in class_names:\n        total = sum(1 for classes in image_class_sets.values() if cls in classes)\n        row = {\"class_name\": cls, \"total_images\": total}\n        for split, ids in split_sets.items():\n            count = sum(1 for image_id in ids if cls in image_class_sets[image_id])\n            row[f\"{split}_images\"] = count\n            row[f\"{split}_pct_of_class\"] = round(100.0 * count / total, 2) if total else 0.0\n        rows.append(row)\n    return pd.DataFrame(rows).sort_values(\"class_name\").reset_index(drop=True)","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:51:44.559927Z","iopub.status.busy":"2026-04-25T06:51:44.559655Z","iopub.status.idle":"2026-04-25T06:51:44.576663Z","shell.execute_reply":"2026-04-25T06:51:44.576061Z"},"papermill":{"duration":0.025479,"end_time":"2026-04-25T06:51:44.578161","exception":false,"start_time":"2026-04-25T06:51:44.552682","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"8753d8e7","cell_type":"code","source":"image_ids_selected = used_sample_ids.copy()\n\nsplit_sets, image_class_sets = build_multilabel_stratified_split(\n    df=df,\n    image_ids=image_ids_selected,\n    split_ratios=SPLIT_RATIOS,\n    seed=SEED,\n)\n\ntrain_ids = split_sets[\"train\"]\nval_ids = split_sets[\"val\"]\ntest_ids = split_sets[\"test\"]\nsplit_distribution_df = summarize_split_distribution(split_sets, image_class_sets)\n\nprint(\"Used images for split:\", len(image_ids_selected))\nprint(\"Split ratios:\", normalize_split_ratios(SPLIT_RATIOS))\nprint(\"Train images:\", len(train_ids))\nprint(\"Val images:\", len(val_ids))\nprint(\"Test images:\", len(test_ids))\nprint(\"Overlap train/val:\", len(train_ids & val_ids))\nprint(\"Overlap train/test:\", len(train_ids & test_ids))\nprint(\"Overlap val/test:\", len(val_ids & test_ids))\n\ndisplay(split_distribution_df)","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:51:44.590713Z","iopub.status.busy":"2026-04-25T06:51:44.590346Z","iopub.status.idle":"2026-04-25T06:51:44.794904Z","shell.execute_reply":"2026-04-25T06:51:44.794091Z"},"papermill":{"duration":0.212594,"end_time":"2026-04-25T06:51:44.796485","exception":false,"start_time":"2026-04-25T06:51:44.583891","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"9a56d612","cell_type":"markdown","source":"# Lưu metadata để lần sau dùng lại","metadata":{"papermill":{"duration":0.00626,"end_time":"2026-04-25T06:51:44.809548","exception":false,"start_time":"2026-04-25T06:51:44.803288","status":"completed"},"tags":[]}},{"id":"ad1bc260","cell_type":"code","source":"import json\n\nmeta = {\n    \"label_csv\": str(LABEL_CSV),\n    \"use_small_subset\": USE_SMALL_SUBSET,\n    \"images_per_class\": IMAGES_PER_CLASS,\n    \"no_finding_images\": NO_FINDING_IMAGES,\n    \"selected_background_ids\": sorted(list(selected_background_ids)),\n    \"seed\": SEED,\n    \"split_ratios\": normalize_split_ratios(SPLIT_RATIOS),\n    \"img_size\": IMG_SIZE,\n    \"save_as_jpg\": SAVE_AS_JPG,\n    \"jpg_quality\": JPG_QUALITY,\n    \"preprocess_combos\": [combo_to_name(combo) for combo in PREPROCESS_COMBOS],\n    \"enable_train_aug\": ENABLE_TRAIN_AUG,\n    \"num_classes\": len(CLASS_NAMES),\n    \"class_names\": CLASS_NAMES,\n    \"wbf_iou_threshold\": WBF_IOU_THRESHOLD,\n    \"wbf_skip_box_threshold\": WBF_SKIP_BOX_THRESHOLD,\n    \"csv_bbox_format\": CSV_BBOX_FORMAT,\n    \"chosen_sample_ids\": sorted(list(chosen_ids)),\n    \"used_sample_ids\": sorted(list(used_sample_ids)),\n    \"missing_dicom_sample_ids\": sorted(list(missing_dicom_ids)),\n    \"train_ids\": sorted(list(train_ids)),\n    \"val_ids\": sorted(list(val_ids)),\n    \"test_ids\": sorted(list(test_ids)),\n}\n\nOUT_ROOT.mkdir(parents=True, exist_ok=True)\nwith open(OUT_ROOT / \"meta.json\", \"w\") as f:\n    json.dump(meta, f, indent=2, ensure_ascii=False)\n\nsplit_distribution_df.to_csv(OUT_ROOT / \"split_distribution.csv\", index=False)\n\nprint(\"Saved:\", OUT_ROOT / \"meta.json\")\nprint(\"Saved:\", OUT_ROOT / \"split_distribution.csv\")","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:51:44.821991Z","iopub.status.busy":"2026-04-25T06:51:44.821572Z","iopub.status.idle":"2026-04-25T06:51:44.845428Z","shell.execute_reply":"2026-04-25T06:51:44.844828Z"},"papermill":{"duration":0.031691,"end_time":"2026-04-25T06:51:44.846754","exception":false,"start_time":"2026-04-25T06:51:44.815063","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"b9ea7f5d","cell_type":"markdown","source":"# Build annotation map nhanh","metadata":{"papermill":{"duration":0.005778,"end_time":"2026-04-25T06:51:44.858733","exception":false,"start_time":"2026-04-25T06:51:44.852955","status":"completed"},"tags":[]}},{"id":"e5a9a525","cell_type":"code","source":"ann_by_sample = {\n    str(sample_id): group.copy()\n    for sample_id, group in df.groupby(\"sample_id\", sort=False)\n}\n\nsplit_map = {}\nfor x in train_ids:\n    split_map[str(x)] = \"train\"\nfor x in val_ids:\n    split_map[str(x)] = \"val\"\nfor x in test_ids:\n    split_map[str(x)] = \"test\"\n\nprint(\"Images with annotations:\", len(ann_by_sample))\nprint(\"Images without annotations:\", len(set(split_map) - set(ann_by_sample)))\nprint(\"Images in split map:\", len(split_map))\nprint(\"Split map counts:\", pd.Series(split_map).value_counts().to_dict())","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:51:44.871082Z","iopub.status.busy":"2026-04-25T06:51:44.870867Z","iopub.status.idle":"2026-04-25T06:51:44.928276Z","shell.execute_reply":"2026-04-25T06:51:44.927437Z"},"papermill":{"duration":0.065623,"end_time":"2026-04-25T06:51:44.929996","exception":false,"start_time":"2026-04-25T06:51:44.864373","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"2009dda4","cell_type":"markdown","source":"# Export ảnh + label nhanh, có track time","metadata":{"papermill":{"duration":0.005671,"end_time":"2026-04-25T06:51:44.94162","exception":false,"start_time":"2026-04-25T06:51:44.935949","status":"completed"},"tags":[]}},{"id":"0934f366","cell_type":"code","source":"import albumentations as A\nimport shutil\nfrom collections import Counter\n\nDICOM_TRAIN_AUG = A.Compose(\n    [\n        A.HorizontalFlip(p=0.5),\n        A.UnsharpMask(\n            blur_limit=(3, 5),\n            sigma_limit=(0.1, 1.5),\n            alpha=(0.1, 0.3),\n            threshold=10,\n            p=0.20,\n        ),\n        A.GaussianBlur(\n            blur_limit=(3, 5),\n            sigma_limit=(0.1, 1.5),\n            p=0.20,\n        ),\n        A.Equalize(\n            mode=\"cv\",\n            by_channels=False,\n            p=0.20,\n        ),\n        A.RandomBrightnessContrast(\n            brightness_limit=0.10,\n            contrast_limit=0.0,\n            p=0.20,\n        ),\n    ],\n    bbox_params=A.BboxParams(\n        format=\"pascal_voc\",\n        label_fields=[\"class_labels\"],\n        min_visibility=0.0,\n    ),\n)\n\n\ndef prepare_combo_output_dirs(combo_name):\n    OUT_ROOT.mkdir(parents=True, exist_ok=True)\n    combo_root = OUT_ROOT / combo_name\n    if REBUILD_OUTPUT and combo_root.exists():\n        shutil.rmtree(combo_root)\n\n    paths = {\n        \"root\": combo_root,\n        \"images\": {},\n        \"labels\": {},\n    }\n    for split in [\"train\", \"val\", \"test\"]:\n        paths[\"images\"][split] = combo_root / \"images\" / split\n        paths[\"labels\"][split] = combo_root / \"labels\" / split\n        paths[\"images\"][split].mkdir(parents=True, exist_ok=True)\n        paths[\"labels\"][split].mkdir(parents=True, exist_ok=True)\n    return paths\n\n\ndef cleanup_combo_dataset(combo_paths):\n    for key in [\"images\", \"labels\"]:\n        root = combo_paths[\"root\"] / key\n        if root.exists():\n            shutil.rmtree(root)\n    print(\"Cleaned combo image/label dirs:\", combo_paths[\"root\"])\n\n\ndef sanitize_pascal_voc_boxes(boxes, width, height):\n    sanitized = []\n    for (x1, y1, x2, y2, class_id) in boxes:\n        x1 = float(np.clip(x1, 0, width))\n        y1 = float(np.clip(y1, 0, height))\n        x2 = float(np.clip(x2, 0, width))\n        y2 = float(np.clip(y2, 0, height))\n\n        if x2 <= x1 or y2 <= y1:\n            continue\n\n        sanitized.append((x1, y1, x2, y2, class_id))\n\n    return sanitized\n\n\ndef apply_dicom_augmentation(img, boxes, split):\n    height, width = img.shape[:2]\n    boxes = sanitize_pascal_voc_boxes(boxes, width=width, height=height)\n\n    if split != \"train\" or len(boxes) == 0 or not ENABLE_TRAIN_AUG:\n        return img, boxes\n\n    bboxes = [(x1, y1, x2, y2) for (x1, y1, x2, y2, _) in boxes]\n    class_labels = [class_id for (_, _, _, _, class_id) in boxes]\n\n    transformed = DICOM_TRAIN_AUG(\n        image=img,\n        bboxes=bboxes,\n        class_labels=class_labels,\n    )\n\n    aug_boxes = [\n        (x1, y1, x2, y2, class_id)\n        for (x1, y1, x2, y2), class_id in zip(\n            transformed[\"bboxes\"],\n            transformed[\"class_labels\"],\n        )\n    ]\n    aug_boxes = sanitize_pascal_voc_boxes(aug_boxes, width=width, height=height)\n    return transformed[\"image\"], aug_boxes\n\n\ndef sample_source_size(rows):\n    if rows is None or len(rows) == 0:\n        return None\n    widths = pd.to_numeric(rows.get(\"source_width\", pd.Series(dtype=float)), errors=\"coerce\").dropna()\n    heights = pd.to_numeric(rows.get(\"source_height\", pd.Series(dtype=float)), errors=\"coerce\").dropna()\n    if len(widths) > 0 and len(heights) > 0 and float(widths.iloc[0]) > 0 and float(heights.iloc[0]) > 0:\n        return float(widths.iloc[0]), float(heights.iloc[0])\n    return None\n\n\ndef detect_bbox_format(rows, source_size, actual_size):\n    if CSV_BBOX_FORMAT != \"auto\":\n        return CSV_BBOX_FORMAT\n    if rows is None or len(rows) == 0:\n        return \"xyxy_pixel\"\n\n    coords = rows[[\"x_min\", \"y_min\", \"x_max\", \"y_max\"]].to_numpy(dtype=float)\n    finite = coords[np.isfinite(coords)]\n    if finite.size and np.nanmax(finite) <= 1.05 and np.nanmin(finite) >= -0.05:\n        return \"xyxy_norm\"\n\n    if source_size is not None:\n        source_w, source_h = source_size\n        max_x = np.nanmax(coords[:, [0, 2]]) if coords.size else 0.0\n        max_y = np.nanmax(coords[:, [1, 3]]) if coords.size else 0.0\n        if max_x > source_w * 1.25 or max_y > source_h * 1.25:\n            raise ValueError(\n                \"CSV bbox lớn hơn nhiều so với source image size tìm được. \"\n                f\"bbox_max=({max_x:.1f}, {max_y:.1f}), source_size=({source_w:.1f}, {source_h:.1f}).\"\n            )\n        return \"xyxy_pixel\"\n\n    actual_w, actual_h = actual_size\n    max_x = np.nanmax(coords[:, [0, 2]]) if coords.size else 0.0\n    max_y = np.nanmax(coords[:, [1, 3]]) if coords.size else 0.0\n    if max_x > actual_w * 1.25 or max_y > actual_h * 1.25:\n        raise ValueError(\n            \"CSV bbox có vẻ là pixel theo ảnh gốc nhưng không khớp DICOM size. \"\n            \"Thêm image_width/image_height vào CSV hoặc kiểm tra bbox source.\"\n        )\n    return \"xyxy_pixel\"\n\n\ndef row_to_yolo_line(row, class2id, source_size, actual_size, bbox_format):\n    cls_idx = class2id[str(row[\"class_name\"])]\n    x1 = float(row[\"x_min\"])\n    y1 = float(row[\"y_min\"])\n    x2 = float(row[\"x_max\"])\n    y2 = float(row[\"y_max\"])\n\n    if bbox_format in {\"xyxy_norm\", \"normalized_xyxy\"}:\n        x1 = float(np.clip(x1, 0.0, 1.0))\n        y1 = float(np.clip(y1, 0.0, 1.0))\n        x2 = float(np.clip(x2, 0.0, 1.0))\n        y2 = float(np.clip(y2, 0.0, 1.0))\n        xc = (x1 + x2) / 2.0\n        yc = (y1 + y2) / 2.0\n        bw = x2 - x1\n        bh = y2 - y1\n    elif bbox_format in {\"xyxy_pixel\", \"pixel_xyxy\"}:\n        denom_w, denom_h = source_size if source_size is not None else actual_size\n        x1 = float(np.clip(x1, 0.0, denom_w))\n        y1 = float(np.clip(y1, 0.0, denom_h))\n        x2 = float(np.clip(x2, 0.0, denom_w))\n        y2 = float(np.clip(y2, 0.0, denom_h))\n        xc = ((x1 + x2) / 2.0) / denom_w\n        yc = ((y1 + y2) / 2.0) / denom_h\n        bw = (x2 - x1) / denom_w\n        bh = (y2 - y1) / denom_h\n    else:\n        raise ValueError(f\"CSV_BBOX_FORMAT không hỗ trợ: {CSV_BBOX_FORMAT}\")\n\n    xc = float(np.clip(xc, 0.0, 1.0))\n    yc = float(np.clip(yc, 0.0, 1.0))\n    bw = float(np.clip(bw, 0.0, 1.0))\n    bh = float(np.clip(bh, 0.0, 1.0))\n    if bw <= 0.0 or bh <= 0.0:\n        return None\n    return f\"{cls_idx} {xc:.6f} {yc:.6f} {bw:.6f} {bh:.6f}\"\n\n\ndef build_yolo_lines_from_rows(rows, class2id, source_size, actual_size):\n    lines = []\n    bbox_format = detect_bbox_format(rows, source_size, actual_size)\n    if rows is None or len(rows) == 0:\n        return lines, bbox_format\n    for _, row in rows.iterrows():\n        line = row_to_yolo_line(row, class2id, source_size, actual_size, bbox_format)\n        if line is not None:\n            lines.append(line)\n    return lines, bbox_format\n\n\ndef rows_to_pascal_voc_boxes(rows, class2id, source_size, actual_size):\n    if rows is None or len(rows) == 0:\n        return []\n\n    bbox_format = detect_bbox_format(rows, source_size, actual_size)\n    actual_w, actual_h = actual_size\n    boxes = []\n\n    for _, row in rows.iterrows():\n        class_id = class2id[str(row[\"class_name\"])]\n        x1 = float(row[\"x_min\"])\n        y1 = float(row[\"y_min\"])\n        x2 = float(row[\"x_max\"])\n        y2 = float(row[\"y_max\"])\n\n        if bbox_format in {\"xyxy_norm\", \"normalized_xyxy\"}:\n            x1 *= actual_w\n            x2 *= actual_w\n            y1 *= actual_h\n            y2 *= actual_h\n        else:\n            denom_w, denom_h = source_size if source_size is not None else actual_size\n            x1 = float(np.clip(x1, 0.0, denom_w)) * actual_w / denom_w\n            x2 = float(np.clip(x2, 0.0, denom_w)) * actual_w / denom_w\n            y1 = float(np.clip(y1, 0.0, denom_h)) * actual_h / denom_h\n            y2 = float(np.clip(y2, 0.0, denom_h)) * actual_h / denom_h\n\n        boxes.append((x1, y1, x2, y2, class_id))\n\n    return sanitize_pascal_voc_boxes(boxes, width=actual_w, height=actual_h)\n\n\ndef build_yolo_lines_from_pascal_boxes(boxes, actual_size):\n    actual_w, actual_h = actual_size\n    lines = []\n    for x1, y1, x2, y2, class_id in boxes:\n        x1 = float(np.clip(x1, 0.0, actual_w))\n        x2 = float(np.clip(x2, 0.0, actual_w))\n        y1 = float(np.clip(y1, 0.0, actual_h))\n        y2 = float(np.clip(y2, 0.0, actual_h))\n        bw = (x2 - x1) / actual_w\n        bh = (y2 - y1) / actual_h\n        if bw <= 0.0 or bh <= 0.0:\n            continue\n        xc = ((x1 + x2) / 2.0) / actual_w\n        yc = ((y1 + y2) / 2.0) / actual_h\n        lines.append(f\"{int(class_id)} {xc:.6f} {yc:.6f} {bw:.6f} {bh:.6f}\")\n    return lines\n\n\ndef write_label_file(label_path: Path, yolo_lines):\n    label_path.parent.mkdir(parents=True, exist_ok=True)\n    label_path.write_text(\"\\n\".join(yolo_lines) + (\"\\n\" if yolo_lines else \"\"))\n\n\ndef save_combo_image_and_label(\n    img,\n    rows,\n    sample_id,\n    split,\n    combo_paths,\n    actual_size,\n    source_size=None,\n    boxes_override=None,\n):\n    img_resized, _, _ = resize_image_keep_shape(img, size=IMG_SIZE)\n    ext = \".jpg\" if SAVE_AS_JPG else \".png\"\n\n    img_out_path = combo_paths[\"images\"][split] / f\"{sample_id}{ext}\"\n    lbl_out_path = combo_paths[\"labels\"][split] / f\"{sample_id}.txt\"\n\n    # Treat stacked methods as RGB-like channels on disk.\n    # OpenCV writes 3-channel arrays as BGR, so reverse before imwrite.\n    img_to_write = img_resized\n    if img_to_write.ndim == 3 and img_to_write.shape[2] == 3:\n        img_to_write = cv2.cvtColor(img_to_write, cv2.COLOR_RGB2BGR)\n\n    if SAVE_AS_JPG:\n        cv2.imwrite(str(img_out_path), img_to_write, [cv2.IMWRITE_JPEG_QUALITY, JPG_QUALITY])\n    else:\n        cv2.imwrite(str(img_out_path), img_to_write, [cv2.IMWRITE_PNG_COMPRESSION, 0])\n\n    if boxes_override is not None:\n        yolo_lines = build_yolo_lines_from_pascal_boxes(boxes_override, actual_size=actual_size)\n        bbox_format = \"aug_pascal_voc\"\n    else:\n        yolo_lines, bbox_format = build_yolo_lines_from_rows(\n            rows=rows,\n            class2id=CLASS2ID,\n            source_size=source_size,\n            actual_size=actual_size,\n        )\n\n    write_label_file(lbl_out_path, yolo_lines)\n    return len(yolo_lines), bbox_format\n\n\ndef process_one_dicom_for_combo(record, combo, combo_paths):\n    t0 = time.perf_counter()\n\n    sample_id = str(record[\"sample_id\"])\n    dicom_path = Path(record[\"path\"])\n    split = split_map.get(sample_id)\n    if split is None:\n        return {\n            \"image_id\": sample_id,\n            \"ok\": False,\n            \"reason\": \"not_in_split\",\n            \"elapsed\": 0.0,\n            \"num_saved\": 0,\n            \"label_count\": 0,\n            \"bbox_format_counts\": {},\n        }\n\n    rows = ann_by_sample.get(sample_id, pd.DataFrame(columns=df.columns))\n\n    try:\n        variants = read_dicom_preprocess_variants(dicom_path)\n        first_variant = next(iter(variants.values()))\n        orig_h, orig_w = first_variant.shape[:2]\n        actual_size = (orig_w, orig_h)\n        source_size = sample_source_size(rows)\n\n        base_boxes = rows_to_pascal_voc_boxes(\n            rows=rows,\n            class2id=CLASS2ID,\n            source_size=source_size,\n            actual_size=actual_size,\n        )\n\n        img_combo = build_preprocess_combo_image(variants, combo)\n        num_saved = 0\n        label_count = 0\n        bbox_format_counts = Counter()\n\n        n_labels, bbox_format = save_combo_image_and_label(\n            img=img_combo,\n            rows=rows,\n            sample_id=sample_id,\n            split=split,\n            combo_paths=combo_paths,\n            actual_size=actual_size,\n            source_size=source_size,\n        )\n        num_saved += 1\n        label_count += n_labels\n        bbox_format_counts[bbox_format] += 1\n\n        if split == \"train\" and len(base_boxes) > 0 and ENABLE_TRAIN_AUG:\n            img_aug, boxes_aug = apply_dicom_augmentation(\n                img_combo.copy(),\n                base_boxes,\n                split,\n            )\n            n_aug_labels, aug_bbox_format = save_combo_image_and_label(\n                img=img_aug,\n                rows=rows,\n                sample_id=f\"{sample_id}_aug\",\n                split=split,\n                combo_paths=combo_paths,\n                actual_size=actual_size,\n                source_size=source_size,\n                boxes_override=boxes_aug,\n            )\n            num_saved += 1\n            label_count += n_aug_labels\n            bbox_format_counts[aug_bbox_format] += 1\n\n        elapsed = time.perf_counter() - t0\n        return {\n            \"image_id\": sample_id,\n            \"ok\": True,\n            \"reason\": \"done\",\n            \"elapsed\": elapsed,\n            \"num_saved\": num_saved,\n            \"label_count\": label_count,\n            \"bbox_format_counts\": dict(bbox_format_counts),\n        }\n\n    except Exception as e:\n        elapsed = time.perf_counter() - t0\n        return {\n            \"image_id\": sample_id,\n            \"ok\": False,\n            \"reason\": str(e),\n            \"elapsed\": elapsed,\n            \"num_saved\": 0,\n            \"label_count\": 0,\n            \"bbox_format_counts\": {},\n        }\n\n\ndef write_combo_yaml(combo_paths):\n    data_yaml = {\n        \"path\": str(combo_paths[\"root\"]),\n        \"train\": \"images/train\",\n        \"val\": \"images/val\",\n        \"test\": \"images/test\",\n        \"names\": {i: name for i, name in enumerate(CLASS_NAMES)},\n    }\n\n    yaml_path = combo_paths[\"root\"] / \"data.yaml\"\n    with open(yaml_path, \"w\") as f:\n        yaml.dump(data_yaml, f, sort_keys=False)\n    return yaml_path\n\n\ndef export_combo_dataset(combo, combo_paths):\n    combo_name = combo_to_name(combo)\n    max_workers = min(8, os.cpu_count() or 4)\n    print(\"max_workers =\", max_workers)\n    print(\"Export combo:\", combo_name)\n\n    total_files = len(dicom_records_selected)\n    start_all = time.perf_counter()\n\n    done = 0\n    ok_count = 0\n    times = []\n    label_count = 0\n    bbox_format_counts = Counter()\n    failed_results = []\n    log_every = 20\n\n    with ThreadPoolExecutor(max_workers=max_workers) as executor:\n        futures = [executor.submit(process_one_dicom_for_combo, record, combo, combo_paths) for record in dicom_records_selected]\n\n        for future in as_completed(futures):\n            result = future.result()\n            done += 1\n\n            if result[\"ok\"]:\n                ok_count += 1\n                times.append(result[\"elapsed\"])\n                label_count += int(result.get(\"label_count\", 0))\n                bbox_format_counts.update(result.get(\"bbox_format_counts\", {}))\n            else:\n                failed_results.append(result)\n\n            if done % log_every == 0 or done == total_files:\n                elapsed_all = time.perf_counter() - start_all\n                avg_per_file = elapsed_all / done\n                speed = done / elapsed_all if elapsed_all > 0 else 0.0\n                remaining = total_files - done\n                eta_sec = remaining * avg_per_file\n                avg_worker = np.mean(times) if len(times) > 0 else 0.0\n                p50_worker = np.median(times) if len(times) > 0 else 0.0\n\n                print(\n                    f\"[{done}/{total_files}] \"\n                    f\"ok={ok_count} | \"\n                    f\"wall={elapsed_all:.1f}s | \"\n                    f\"speed={speed:.2f} dicom/s | \"\n                    f\"avg_wall/dicom={avg_per_file:.3f}s | \"\n                    f\"avg_worker={avg_worker:.3f}s | \"\n                    f\"p50_worker={p50_worker:.3f}s | \"\n                    f\"ETA={eta_sec:.1f}s\"\n                )\n\n    total_elapsed = time.perf_counter() - start_all\n    img_ext = \"*.jpg\" if SAVE_AS_JPG else \"*.png\"\n    split_image_counts = {\n        split: len(list(combo_paths[\"images\"][split].glob(img_ext)))\n        for split in [\"train\", \"val\", \"test\"]\n    }\n    split_label_counts = {\n        split: len(list(combo_paths[\"labels\"][split].glob(\"*.txt\")))\n        for split in [\"train\", \"val\", \"test\"]\n    }\n\n    stats = {\n        \"combo\": combo_name,\n        \"export_processed_dicom\": done,\n        \"export_success_dicom\": ok_count,\n        \"export_failed_dicom\": len(failed_results),\n        \"export_seconds\": total_elapsed,\n        \"export_label_count\": label_count,\n        \"bbox_format_counts\": dict(bbox_format_counts),\n        \"train_images_exported\": split_image_counts[\"train\"],\n        \"val_images_exported\": split_image_counts[\"val\"],\n        \"test_images_exported\": split_image_counts[\"test\"],\n        \"train_labels_exported\": split_label_counts[\"train\"],\n        \"val_labels_exported\": split_label_counts[\"val\"],\n        \"test_labels_exported\": split_label_counts[\"test\"],\n    }\n    if failed_results:\n        print(\"First failed results:\")\n        display(pd.DataFrame(failed_results).head(10))\n    return stats","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:51:44.955211Z","iopub.status.busy":"2026-04-25T06:51:44.954761Z","iopub.status.idle":"2026-04-25T06:51:44.976571Z","shell.execute_reply":"2026-04-25T06:51:44.976007Z"},"papermill":{"duration":0.030756,"end_time":"2026-04-25T06:51:44.978077","exception":false,"start_time":"2026-04-25T06:51:44.947321","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"6d48bb13","cell_type":"markdown","source":"# Chạy export song song + ETA","metadata":{"papermill":{"duration":0.005923,"end_time":"2026-04-25T06:51:44.989896","exception":false,"start_time":"2026-04-25T06:51:44.983973","status":"completed"},"tags":[]}},{"id":"863c2b55","cell_type":"code","source":"import torch\nfrom ultralytics import YOLO\n\nDEVICE = 0 if torch.cuda.is_available() else \"cpu\"\nprint(\"Device:\", DEVICE)\n\ncombo_pipeline_rows = []\ntrained_run_info = {}\nsummary_path = OUT_ROOT / \"sequential_combo_results.csv\"\nOUT_ROOT.mkdir(parents=True, exist_ok=True)\n\nfor combo in PREPROCESS_COMBOS:\n    combo_name = combo_to_name(combo)\n    combo_paths = prepare_combo_output_dirs(combo_name)\n    yaml_path = None\n    row = {\n        \"combo\": combo_name,\n        \"status\": \"started\",\n    }\n\n    print(\"\\n\" + \"=\" * 100)\n    print(\"Combo:\", combo_name)\n    print(\"Dataset root:\", combo_paths[\"root\"])\n\n    try:\n        export_stats = export_combo_dataset(combo, combo_paths)\n        row.update(export_stats)\n\n        yaml_path = write_combo_yaml(combo_paths)\n        row[\"yaml_path\"] = str(yaml_path)\n        print(\"Saved YAML:\", yaml_path)\n\n        run_name = f\"{Path(MODEL_NAME).stem}_{combo_name}\"\n        row[\"run_name\"] = run_name\n        print(\"Training run:\", run_name)\n\n        model = YOLO(MODEL_NAME)\n        model.train(\n            data=str(yaml_path),\n            epochs=TRAIN_EPOCHS,\n            imgsz=IMG_SIZE,\n            batch=TRAIN_BATCH,\n            device=DEVICE,\n            project=str(OUT_ROOT / \"runs\"),\n            name=run_name,\n            exist_ok=True,\n        )\n\n        best_weight_path = OUT_ROOT / \"runs\" / run_name / \"weights\" / \"best.pt\"\n        row[\"best_weight_path\"] = str(best_weight_path)\n        trained_run_info[combo_name] = {\n            \"run_name\": run_name,\n            \"yaml_path\": str(yaml_path),\n            \"best_weight_path\": str(best_weight_path),\n        }\n\n        if not best_weight_path.exists():\n            raise FileNotFoundError(f\"Không thấy best weight sau train: {best_weight_path}\")\n\n        print(\"Testing combo:\", combo_name)\n        best_model = YOLO(str(best_weight_path))\n        metrics = best_model.val(\n            data=str(yaml_path),\n            split=\"test\",\n            imgsz=IMG_SIZE,\n            batch=EVAL_BATCH,\n            device=DEVICE,\n            verbose=True,\n        )\n\n        row.update({\n            \"precision\": float(metrics.box.mp),\n            \"recall\": float(metrics.box.mr),\n            \"map50\": float(metrics.box.map50),\n            \"map50_95\": float(metrics.box.map),\n            \"status\": \"done\",\n        })\n\n    except Exception as e:\n        row[\"status\"] = \"failed\"\n        row[\"error\"] = str(e)\n        print(\"Combo failed:\", combo_name)\n        print(e)\n\n    finally:\n        combo_pipeline_rows.append(row)\n        combo_pipeline_results_df = pd.DataFrame(combo_pipeline_rows)\n        combo_pipeline_results_df.to_csv(summary_path, index=False)\n        print(\"Saved running summary:\", summary_path)\n        display(combo_pipeline_results_df.tail(1))\n\n        if CLEANUP_AFTER_COMBO:\n            cleanup_combo_dataset(combo_paths)\n\n    if row.get(\"status\") != \"done\":\n        raise RuntimeError(f\"Dừng pipeline vì combo failed: {combo_name}\")\n\nprint(\"\\nAll combos finished\")\nprint(\"Saved summary:\", summary_path)\ndisplay(combo_pipeline_results_df)","metadata":{"execution":{"iopub.execute_input":"2026-04-25T06:51:45.00286Z","iopub.status.busy":"2026-04-25T06:51:45.00248Z","iopub.status.idle":"2026-04-25T07:41:01.846281Z","shell.execute_reply":"2026-04-25T07:41:01.845254Z"},"papermill":{"duration":2956.852778,"end_time":"2026-04-25T07:41:01.848573","exception":false,"start_time":"2026-04-25T06:51:44.995795","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"7c681abe","cell_type":"code","source":"print(\"Sequential combo pipeline summary\")\nprint(\"Summary CSV:\", OUT_ROOT / \"sequential_combo_results.csv\")\ndisplay(combo_pipeline_results_df)","metadata":{"execution":{"iopub.execute_input":"2026-04-25T07:41:01.886575Z","iopub.status.busy":"2026-04-25T07:41:01.885921Z","iopub.status.idle":"2026-04-25T07:41:01.952291Z","shell.execute_reply":"2026-04-25T07:41:01.951408Z"},"papermill":{"duration":0.085881,"end_time":"2026-04-25T07:41:01.953745","exception":false,"start_time":"2026-04-25T07:41:01.867864","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"93abd1ec","cell_type":"markdown","source":"# Tạo data.yaml","metadata":{"papermill":{"duration":0.016844,"end_time":"2026-04-25T07:41:01.988464","exception":false,"start_time":"2026-04-25T07:41:01.97162","status":"completed"},"tags":[]}},{"id":"63016dc7","cell_type":"code","source":"# data.yaml được tạo tạm cho từng combo trong sequential pipeline cell.\n# Sau khi train/eval xong, images/labels của combo được xóa nếu CLEANUP_AFTER_COMBO=True.\nprint(\"YAML creation is handled inside the sequential combo pipeline.\")\nprint(\"CLEANUP_AFTER_COMBO:\", CLEANUP_AFTER_COMBO)","metadata":{"execution":{"iopub.execute_input":"2026-04-25T07:41:02.023867Z","iopub.status.busy":"2026-04-25T07:41:02.023111Z","iopub.status.idle":"2026-04-25T07:41:02.029798Z","shell.execute_reply":"2026-04-25T07:41:02.029136Z"},"papermill":{"duration":0.025958,"end_time":"2026-04-25T07:41:02.031168","exception":false,"start_time":"2026-04-25T07:41:02.00521","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"9e26e4ec","cell_type":"code","source":"# Preview chỉ dùng được nếu CLEANUP_AFTER_COMBO=False hoặc chạy riêng một combo và chưa cleanup.\nprint(\"Preview skipped because sequential mode cleans combo images/labels to save Kaggle disk.\")\nprint(\"Set CLEANUP_AFTER_COMBO=False if you need to preview exported images before cleanup.\")","metadata":{"execution":{"iopub.execute_input":"2026-04-25T07:41:02.066868Z","iopub.status.busy":"2026-04-25T07:41:02.066574Z","iopub.status.idle":"2026-04-25T07:41:05.173766Z","shell.execute_reply":"2026-04-25T07:41:05.172791Z"},"papermill":{"duration":3.130848,"end_time":"2026-04-25T07:41:05.17939","exception":false,"start_time":"2026-04-25T07:41:02.048542","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"2d547422","cell_type":"code","source":"# Training đã được xử lý trong sequential combo pipeline cell.\nprint(\"Training already completed in the sequential combo pipeline cell.\")\ndisplay(combo_pipeline_results_df)","metadata":{"execution":{"iopub.execute_input":"2026-04-25T07:41:05.29518Z","iopub.status.busy":"2026-04-25T07:41:05.294553Z","iopub.status.idle":"2026-04-25T08:57:45.411746Z","shell.execute_reply":"2026-04-25T08:57:45.410936Z"},"papermill":{"duration":4601.535779,"end_time":"2026-04-25T08:57:46.771469","exception":false,"start_time":"2026-04-25T07:41:05.23569","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"f216fb7c","cell_type":"code","source":"# Eval metrics đã được xử lý trong sequential combo pipeline cell.\nmetrics_path = OUT_ROOT / \"sequential_combo_results.csv\"\nprint(\"Saved metrics:\", metrics_path)\ndisplay(pd.read_csv(metrics_path))","metadata":{"execution":{"iopub.execute_input":"2026-04-25T08:57:49.166431Z","iopub.status.busy":"2026-04-25T08:57:49.165644Z","iopub.status.idle":"2026-04-25T08:58:02.741108Z","shell.execute_reply":"2026-04-25T08:58:02.740049Z"},"papermill":{"duration":14.728231,"end_time":"2026-04-25T08:58:02.742717","exception":false,"start_time":"2026-04-25T08:57:48.014486","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}