{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":15094918,"datasetId":9664561,"databundleVersionId":15979687}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":18220.537857,"end_time":"2026-03-28T06:09:11.111189+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-03-28T01:05:30.573332+00:00","version":"2.7.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"eb19237f","cell_type":"code","source":"!pip -q install pydicom pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg ultralytics","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2026-03-28T01:05:32.937439Z","iopub.status.busy":"2026-03-28T01:05:32.937092Z","iopub.status.idle":"2026-03-28T01:05:39.302184Z","shell.execute_reply":"2026-03-28T01:05:39.301418Z"},"papermill":{"duration":6.372343,"end_time":"2026-03-28T01:05:39.303966+00:00","exception":false,"start_time":"2026-03-28T01:05:32.931623+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"f8c24f41","cell_type":"code","source":"import os\nimport cv2\nimport yaml\nimport time\nimport random\nimport numpy as np\nimport pandas as pd\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_modality_lut, apply_voi_lut\n\nfrom collections import defaultdict\nfrom sklearn.model_selection import train_test_split\nfrom concurrent.futures import ThreadPoolExecutor, as_completed","metadata":{"execution":{"iopub.execute_input":"2026-03-28T01:05:39.314438Z","iopub.status.busy":"2026-03-28T01:05:39.313529Z","iopub.status.idle":"2026-03-28T01:05:42.675002Z","shell.execute_reply":"2026-03-28T01:05:42.674403Z"},"papermill":{"duration":3.368333,"end_time":"2026-03-28T01:05:42.67678+00:00","exception":false,"start_time":"2026-03-28T01:05:39.308447+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"dee625b9","cell_type":"markdown","source":"# Config","metadata":{"papermill":{"duration":0.003999,"end_time":"2026-03-28T01:05:42.685057+00:00","exception":false,"start_time":"2026-03-28T01:05:42.681058+00:00","status":"completed"},"tags":[]}},{"id":"4c39f19e","cell_type":"code","source":"# =========================\n# PATH CONFIG\n# =========================\nfrom pathlib import Path\n\nROOT = Path(\"/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection\")\n\nTRAIN_DIR = ROOT / \"train\"\nTRAIN_CSV = '/kaggle/input/datasets/benxelua/correct-label/annotations/annotations_train.csv'\n\n# output YOLO dataset\nOUT_ROOT = Path(\"/kaggle/working/yolo_vindr_multiclass\")\n\nPNG_DIR = OUT_ROOT / \"images\"\nLBL_DIR = OUT_ROOT / \"labels\"\n\nIMG_TRAIN_DIR = PNG_DIR / \"train\"\nIMG_VAL_DIR   = PNG_DIR / \"val\"\nLBL_TRAIN_DIR = LBL_DIR / \"train\"\nLBL_VAL_DIR   = LBL_DIR / \"val\"\n\nfor d in [IMG_TRAIN_DIR, IMG_VAL_DIR, LBL_TRAIN_DIR, LBL_VAL_DIR]:\n    d.mkdir(parents=True, exist_ok=True)\n\n# =========================\n# SUBSET CONFIG\n# =========================\nUSE_SMALL_SUBSET = False     # True = subset nhỏ, False = full dataset\nIMAGES_PER_CLASS = 12       # mỗi class lấy khoảng 10-15 ảnh, ví dụ 12\nSEED = 42\n\n# =========================\n# IMAGE CONFIG\n# =========================\nIMG_SIZE = 1024             # có thể đổi 640 cho nhanh hơn\nSAVE_AS_JPG = False         # True nhanh hơn PNG\nJPG_QUALITY = 95\nENABLE_AUGMENTATION = True  # Albumentations Dataset Duplication Enabled\n","metadata":{"execution":{"iopub.execute_input":"2026-03-28T01:05:42.694893Z","iopub.status.busy":"2026-03-28T01:05:42.694586Z","iopub.status.idle":"2026-03-28T01:05:42.70096Z","shell.execute_reply":"2026-03-28T01:05:42.700441Z"},"papermill":{"duration":0.012574,"end_time":"2026-03-28T01:05:42.702496+00:00","exception":false,"start_time":"2026-03-28T01:05:42.689922+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"8c33ec3c","cell_type":"markdown","source":"# Helper functions","metadata":{"papermill":{"duration":0.003808,"end_time":"2026-03-28T01:05:42.710223+00:00","exception":false,"start_time":"2026-03-28T01:05:42.706415+00:00","status":"completed"},"tags":[]}},{"id":"12a498b5","cell_type":"code","source":"import random\nimport cv2\nimport numpy as np\nimport pydicom\n\ndef seed_everything(seed=42):\n    random.seed(seed)\n    np.random.seed(seed)\n\nseed_everything(42) # Reverting to hardcoded seed 42 to avoid NameError on SEED if not defined earlier\n\ndef list_dicom_files(folder: Path):\n    files = []\n    for p in sorted(folder.rglob(\"*\")):\n        if p.is_file() and p.suffix.lower() in {\".dicom\", \".dcm\"}:\n            files.append(p)\n    return files\n\ndef resize_image_keep_shape(img, size=1024):\n    h, w = img.shape[:2]\n    resized = cv2.resize(img, (size, size), interpolation=cv2.INTER_AREA)\n    return resized, w, h\n\nfrom pydicom.pixel_data_handlers.util import apply_modality_lut, apply_voi_lut\n\ndef read_dicom_to_uint8(path: Path):\n    ds = pydicom.dcmread(str(path))\n    \n    # 1. Apply Modality LUT\n    img = apply_modality_lut(ds.pixel_array, ds)\n    \n    # 2. Apply VOI LUT (Windowing)\n    img = apply_voi_lut(img, ds)\n    \n    # 3. Handle Photometric Interpretation (Invert MONOCHROME1)\n    if getattr(ds, \"PhotometricInterpretation\", \"\") == \"MONOCHROME1\":\n        img = np.amax(img) - img\n\n    # 4. Normalize cleanly to 8-bit (0-255) uint8\n    img = img - np.min(img)\n    if np.max(img) > 0:\n        img = img / np.max(img)\n    \n    img = (img * 255.0).clip(0, 255).astype(np.uint8)\n    return img\n","metadata":{"execution":{"iopub.execute_input":"2026-03-28T01:05:42.719514Z","iopub.status.busy":"2026-03-28T01:05:42.71884Z","iopub.status.idle":"2026-03-28T01:05:42.725852Z","shell.execute_reply":"2026-03-28T01:05:42.725311Z"},"papermill":{"duration":0.013256,"end_time":"2026-03-28T01:05:42.727241+00:00","exception":false,"start_time":"2026-03-28T01:05:42.713985+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"61e837d6","cell_type":"markdown","source":"# Đọc CSV và chuẩn hóa annotation","metadata":{"papermill":{"duration":0.003688,"end_time":"2026-03-28T01:05:42.734738+00:00","exception":false,"start_time":"2026-03-28T01:05:42.73105+00:00","status":"completed"},"tags":[]}},{"id":"4f72a3fd","cell_type":"code","source":"df_full = pd.read_csv(TRAIN_CSV)\n\nprint(\"Full CSV shape:\", df_full.shape)\nprint(\"Columns:\", df_full.columns.tolist())\ndisplay(df_full.head())\n\n# Bỏ No finding\nif \"class_name\" in df_full.columns:\n    df_full = df_full[df_full[\"class_name\"].fillna(\"\").str.lower() != \"no finding\"].copy()\nelse:\n    raise ValueError(\"train.csv của bạn cần có cột class_name để làm multi-class dễ hơn.\")\n\n# Giữ bbox hợp lệ\nrequired_cols = [\"image_id\", \"class_name\", \"x_min\", \"y_min\", \"x_max\", \"y_max\"]\nfor c in required_cols:\n    if c not in df_full.columns:\n        raise ValueError(f\"Thiếu cột {c}\")\n\ndf_full = df_full.dropna(subset=required_cols).copy()\ndf_full = df_full[(df_full[\"x_max\"] > df_full[\"x_min\"]) & (df_full[\"y_max\"] > df_full[\"y_min\"])].copy()\n\n# Build class list\nCLASS_NAMES = sorted(df_full[\"class_name\"].unique().tolist())\nCLASS2ID = {name: i for i, name in enumerate(CLASS_NAMES)}\nID2CLASS = {i: name for name, i in CLASS2ID.items()}\n\nprint(\"Num classes:\", len(CLASS_NAMES))\nprint(\"CLASS_NAMES:\", CLASS_NAMES)\nprint(\"CLASS2ID:\", CLASS2ID)","metadata":{"execution":{"iopub.execute_input":"2026-03-28T01:05:42.743697Z","iopub.status.busy":"2026-03-28T01:05:42.743199Z","iopub.status.idle":"2026-03-28T01:05:42.994087Z","shell.execute_reply":"2026-03-28T01:05:42.99338Z"},"papermill":{"duration":0.256923,"end_time":"2026-03-28T01:05:42.995556+00:00","exception":false,"start_time":"2026-03-28T01:05:42.738633+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"9b442b77","cell_type":"markdown","source":"# Chọn subset nhỏ theo từng class","metadata":{"papermill":{"duration":0.00415,"end_time":"2026-03-28T01:05:43.004202+00:00","exception":false,"start_time":"2026-03-28T01:05:43.000052+00:00","status":"completed"},"tags":[]}},{"id":"f222cd85","cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nprint(\"Building subset from CSV only...\")\n\n# chỉ dùng annotation đã clean\ndf_work = df_full.copy()\n\n# group 1 lần\nclass_to_ids = (\n    df_work.groupby(\"class_name\")[\"image_id\"]\n    .unique()\n    .to_dict()\n)\n\ndef build_small_subset_by_class_fast(class_to_ids, images_per_class=12, seed=42):\n    rng = np.random.default_rng(seed)\n\n    selected_ids = set()\n    per_class_summary = []\n\n    for cls in sorted(class_to_ids.keys()):\n        cls_ids = list(class_to_ids[cls])\n        n_take = min(images_per_class, len(cls_ids))\n\n        if n_take > 0:\n            chosen = rng.choice(cls_ids, size=n_take, replace=False).tolist()\n        else:\n            chosen = []\n\n        selected_ids.update(chosen)\n\n        per_class_summary.append({\n            \"class_name\": cls,\n            \"available_images\": len(cls_ids),\n            \"selected_images\": len(chosen)\n        })\n\n    return selected_ids, pd.DataFrame(per_class_summary)\n\ndicom_files_all = list_dicom_files(TRAIN_DIR)\n\nall_file_ids = {p.stem for p in dicom_files_all}\n\nif USE_SMALL_SUBSET:\n    chosen_ids, subset_summary_df = build_small_subset_by_class_fast(\n        class_to_ids,\n        images_per_class=IMAGES_PER_CLASS,\n        seed=SEED\n    )\n    print(\"Using SMALL subset mode\")\n    print(\"Total selected unique images:\", len(chosen_ids))\n    display(subset_summary_df)\nelse:\n    chosen_ids = set(all_file_ids)\n    print(\"Using FULL dataset mode\")\n    print(\"Total selected unique images:\", len(chosen_ids))\n\n\nprint(\"Scanning DICOM files...\")\n\n# chỉ giữ file thuộc chosen_ids\ndicom_files_selected = [p for p in dicom_files_all if p.stem in chosen_ids]\n\nprint(\"Total DICOM files scanned:\", len(dicom_files_all))\nprint(\"Selected DICOM files:\", len(dicom_files_selected))","metadata":{"execution":{"iopub.execute_input":"2026-03-28T01:05:43.013821Z","iopub.status.busy":"2026-03-28T01:05:43.01315Z","iopub.status.idle":"2026-03-28T01:06:14.740771Z","shell.execute_reply":"2026-03-28T01:06:14.739928Z"},"papermill":{"duration":31.737887,"end_time":"2026-03-28T01:06:14.746064+00:00","exception":false,"start_time":"2026-03-28T01:05:43.008177+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"d6642705","cell_type":"code","source":"dicom_files_selected = [p for p in dicom_files_all if p.stem in chosen_ids]\ndf = df_full[df_full[\"image_id\"].isin(chosen_ids)].copy()\n\nprint(\"Selected DICOM files:\", len(dicom_files_selected))\nprint(\"Selected annotation rows:\", len(df))\nprint(\"Selected unique image_id:\", df[\"image_id\"].nunique())","metadata":{"execution":{"iopub.execute_input":"2026-03-28T01:06:14.755685Z","iopub.status.busy":"2026-03-28T01:06:14.75546Z","iopub.status.idle":"2026-03-28T01:06:14.785329Z","shell.execute_reply":"2026-03-28T01:06:14.784489Z"},"papermill":{"duration":0.03632,"end_time":"2026-03-28T01:06:14.786712+00:00","exception":false,"start_time":"2026-03-28T01:06:14.750392+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"e9a4b8a7","cell_type":"markdown","source":"# Split train/val","metadata":{"papermill":{"duration":0.004051,"end_time":"2026-03-28T01:06:14.795232+00:00","exception":false,"start_time":"2026-03-28T01:06:14.791181+00:00","status":"completed"},"tags":[]}},{"id":"f3f7c625","cell_type":"code","source":"from collections import defaultdict\nimport numpy as np\n\ndef build_multilabel_split(df, image_ids, val_ratio=0.2, seed=42):\n    rng = np.random.default_rng(seed)\n\n    image_ids = list(sorted(set(image_ids)))\n    class_to_images = {\n        cls: set(df.loc[df[\"class_name\"] == cls, \"image_id\"].unique()) & set(image_ids)\n        for cls in sorted(df[\"class_name\"].unique())\n    }\n\n    val_ids = set()\n\n    # Mỗi class cố gắng lấy ít nhất 1 ảnh vào val\n    for cls, ids in class_to_images.items():\n        ids = list(ids - val_ids)\n        if len(ids) > 0:\n            chosen = rng.choice(ids, size=1, replace=False)\n            val_ids.update(chosen.tolist())\n\n    # Bổ sung cho đủ tỷ lệ val mong muốn\n    target_val_size = max(1, int(len(image_ids) * val_ratio))\n    remaining = list(set(image_ids) - val_ids)\n\n    if len(val_ids) < target_val_size and len(remaining) > 0:\n        n_more = min(target_val_size - len(val_ids), len(remaining))\n        extra = rng.choice(remaining, size=n_more, replace=False)\n        val_ids.update(extra.tolist())\n\n    train_ids = set(image_ids) - val_ids\n    return train_ids, val_ids","metadata":{"execution":{"iopub.execute_input":"2026-03-28T01:06:14.804607Z","iopub.status.busy":"2026-03-28T01:06:14.804391Z","iopub.status.idle":"2026-03-28T01:06:14.81071Z","shell.execute_reply":"2026-03-28T01:06:14.810176Z"},"papermill":{"duration":0.012572,"end_time":"2026-03-28T01:06:14.812036+00:00","exception":false,"start_time":"2026-03-28T01:06:14.799464+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"b7d15a01","cell_type":"code","source":"image_ids_selected = sorted([p.stem for p in dicom_files_selected])\n\ntrain_ids, val_ids = build_multilabel_split(\n    df=df,\n    image_ids=image_ids_selected,\n    val_ratio=0.2,\n    seed=SEED\n)\n\nprint(\"Train images:\", len(train_ids))\nprint(\"Val images:\", len(val_ids))","metadata":{"execution":{"iopub.execute_input":"2026-03-28T01:06:14.821276Z","iopub.status.busy":"2026-03-28T01:06:14.821021Z","iopub.status.idle":"2026-03-28T01:06:14.93405Z","shell.execute_reply":"2026-03-28T01:06:14.933311Z"},"papermill":{"duration":0.119342,"end_time":"2026-03-28T01:06:14.935548+00:00","exception":false,"start_time":"2026-03-28T01:06:14.816206+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"d647772e","cell_type":"markdown","source":"# Lưu metadata để lần sau dùng lại","metadata":{"papermill":{"duration":0.004301,"end_time":"2026-03-28T01:06:14.944525+00:00","exception":false,"start_time":"2026-03-28T01:06:14.940224+00:00","status":"completed"},"tags":[]}},{"id":"bebe7a09","cell_type":"code","source":"import json\n\nmeta = {\n    \"use_small_subset\": USE_SMALL_SUBSET,\n    \"images_per_class\": IMAGES_PER_CLASS,\n    \"seed\": SEED,\n    \"img_size\": IMG_SIZE,\n    \"save_as_jpg\": SAVE_AS_JPG,\n    \"jpg_quality\": JPG_QUALITY,\n    \"num_classes\": len(CLASS_NAMES),\n    \"class_names\": CLASS_NAMES,\n    \"chosen_ids\": sorted(list(chosen_ids)),\n    \"train_ids\": sorted(list(train_ids)),\n    \"val_ids\": sorted(list(val_ids)),\n}\n\nwith open(OUT_ROOT / \"meta.json\", \"w\") as f:\n    json.dump(meta, f, indent=2)\n\nprint(\"Saved:\", OUT_ROOT / \"meta.json\")","metadata":{"execution":{"iopub.execute_input":"2026-03-28T01:06:14.955017Z","iopub.status.busy":"2026-03-28T01:06:14.954309Z","iopub.status.idle":"2026-03-28T01:06:14.97844Z","shell.execute_reply":"2026-03-28T01:06:14.977763Z"},"papermill":{"duration":0.030466,"end_time":"2026-03-28T01:06:14.979823+00:00","exception":false,"start_time":"2026-03-28T01:06:14.949357+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"b1768ad9","cell_type":"markdown","source":"# Build annotation map nhanh","metadata":{"papermill":{"duration":0.004469,"end_time":"2026-03-28T01:06:14.988631+00:00","exception":false,"start_time":"2026-03-28T01:06:14.984162+00:00","status":"completed"},"tags":[]}},{"id":"57164b4d","cell_type":"code","source":"from collections import defaultdict\n\nann_map = defaultdict(list)\n\nfor r in df.itertuples(index=False):\n    class_id = CLASS2ID[r.class_name]\n    ann_map[r.image_id].append(\n        (r.x_min, r.y_min, r.x_max, r.y_max, class_id)\n    )\n\nsplit_map = {}\nfor x in train_ids:\n    split_map[x] = \"train\"\nfor x in val_ids:\n    split_map[x] = \"val\"\n\nprint(\"Images with annotations:\", len(ann_map))\nprint(\"Images in split map:\", len(split_map))","metadata":{"execution":{"iopub.execute_input":"2026-03-28T01:06:14.99888Z","iopub.status.busy":"2026-03-28T01:06:14.9986Z","iopub.status.idle":"2026-03-28T01:06:15.153609Z","shell.execute_reply":"2026-03-28T01:06:15.152663Z"},"papermill":{"duration":0.161328,"end_time":"2026-03-28T01:06:15.154968+00:00","exception":false,"start_time":"2026-03-28T01:06:14.99364+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"da7301ec","cell_type":"markdown","source":"# Export ảnh + label nhanh, có track time","metadata":{"papermill":{"duration":0.004331,"end_time":"2026-03-28T01:06:15.163983+00:00","exception":false,"start_time":"2026-03-28T01:06:15.159652+00:00","status":"completed"},"tags":[]}},{"id":"b7be58e2","cell_type":"code","source":"import albumentations as A\n\n# Safe clinical parameters explicitly authorized by user research\nDICOM_TRAIN_AUG = A.Compose(\n    [\n        A.HorizontalFlip(p=0.5),\n        A.UnsharpMask(blur_limit=(3, 5), sigma_limit=(0.1, 1.5), alpha=(0.1, 0.3), threshold=10, p=0.20),\n        A.GaussianBlur(blur_limit=(3, 5), sigma_limit=(0.1, 1.5), p=0.20)\n    ],\n    bbox_params=A.BboxParams(\n        format=\"pascal_voc\",\n        label_fields=[\"class_labels\"],\n        min_visibility=0.0,\n    ),\n)\n\ndef sanitize_pascal_voc_boxes(boxes_scaled, width, height):\n    sanitized = []\n    for (x1, y1, x2, y2, class_id) in boxes_scaled:\n        x1 = float(np.clip(x1, 0, width))\n        y1 = float(np.clip(y1, 0, height))\n        x2 = float(np.clip(x2, 0, width))\n        y2 = float(np.clip(y2, 0, height))\n        if x2 <= x1 or y2 <= y1: continue\n        sanitized.append((x1, y1, x2, y2, class_id))\n    return sanitized\n\ndef apply_dicom_augmentation(img_scaled, boxes_scaled, split):\n    # Both img_scaled and boxes_scaled are natively IMG_SIZE x IMG_SIZE natively\n    height, width = img_scaled.shape[:2]\n    boxes_scaled = sanitize_pascal_voc_boxes(boxes_scaled, width=width, height=height)\n\n    if split != \"train\" or len(boxes_scaled) == 0:\n        return img_scaled, boxes_scaled\n\n    bboxes = [(x1, y1, x2, y2) for (x1, y1, x2, y2, _) in boxes_scaled]\n    class_labels = [class_id for (_, _, _, _, class_id) in boxes_scaled]\n\n    transformed = DICOM_TRAIN_AUG(\n        image=img_scaled,\n        bboxes=bboxes,\n        class_labels=class_labels,\n    )\n\n    aug_boxes = [\n        (x1, y1, x2, y2, class_id)\n        for (x1, y1, x2, y2), class_id in zip(transformed[\"bboxes\"], transformed[\"class_labels\"])\n    ]\n    return transformed[\"image\"], sanitize_pascal_voc_boxes(aug_boxes, width=width, height=height)\n\ndef build_yolo_lines(boxes_scaled):\n    yolo_lines = []\n    # Because boxes are natively bound inside IMG_SIZE x IMG_SIZE natively...\n    for (x1, y1, x2, y2, class_id) in boxes_scaled:\n        xc = ((x1 + x2) / 2.0) / IMG_SIZE\n        yc = ((y1 + y2) / 2.0) / IMG_SIZE\n        bw = (x2 - x1) / IMG_SIZE\n        bh = (y2 - y1) / IMG_SIZE\n\n        xc = min(max(xc, 0.0), 1.0)\n        yc = min(max(yc, 0.0), 1.0)\n        bw = min(max(bw, 0.0), 1.0)\n        bh = min(max(bh, 0.0), 1.0)\n\n        if bw > 0 and bh > 0:\n            yolo_lines.append(f\"{class_id} {xc:.6f} {yc:.6f} {bw:.6f} {bh:.6f}\")\n    return yolo_lines\n\ndef save_image_and_label(img_ready, boxes_scaled, image_id, split):\n    ext = \".jpg\" if SAVE_AS_JPG else \".png\"\n\n    if split == \"train\":\n        img_out_path = IMG_TRAIN_DIR / f\"{image_id}{ext}\"\n        lbl_out_path = LBL_TRAIN_DIR / f\"{image_id}.txt\"\n    else:\n        img_out_path = IMG_VAL_DIR / f\"{image_id}{ext}\"\n        lbl_out_path = LBL_VAL_DIR / f\"{image_id}.txt\"\n\n        \n    # KAGGLE FAILSAFE DISK COMPRESSION BOUNDS\n    if SAVE_AS_JPG:\n        cv2.imwrite(str(img_out_path), img_ready, [cv2.IMWRITE_JPEG_QUALITY, 95])\n    else:\n        cv2.imwrite(str(img_out_path), img_ready, [cv2.IMWRITE_PNG_COMPRESSION, 9]) \n\n    if len(boxes_scaled) > 0 or split == \"val\":\n        yolo_lines = build_yolo_lines(boxes_scaled)\n        with open(lbl_out_path, \"w\") as f:\n            f.write(\"\\n\".join(yolo_lines))\n\ndef process_one_dicom(dicom_path: Path):\n    t0 = time.perf_counter()\n\n    image_id = dicom_path.stem\n    split = split_map.get(image_id)\n    if split is None:\n        return {\"image_id\": image_id, \"ok\": False, \"reason\": \"not_in_split\", \"elapsed\": 0.0, \"num_saved\": 0}\n\n    try:\n        # Extract 3000x3000 full raw matrix\n        img_raw = read_dicom_to_uint8(dicom_path)\n        orig_h, orig_w = img_raw.shape[:2]\n        \n        # 1. KAGGLE FREEZE FIX: Scale aggressively to IMG_SIZE BEFORE computationally expensive steps\n        img_resized = cv2.resize(img_raw, (IMG_SIZE, IMG_SIZE), interpolation=cv2.INTER_AREA)\n        \n        # Scale boxes proportionately mathematically from original to IMG_SIZE bounds\n        boxes = ann_map.get(image_id, [])\n        scaled_boxes = []\n        if len(boxes) > 0:\n            sx = IMG_SIZE / orig_w\n            sy = IMG_SIZE / orig_h\n            for (x1, y1, x2, y2, class_id) in boxes:\n                 scaled_boxes.append((x1*sx, y1*sy, x2*sx, y2*sy, class_id))\n\n        # A. ALWAYS SAVE PRISTINE ORIGINAL\n        save_image_and_label(img_resized, scaled_boxes, image_id, split)\n        num_saved = 1\n\n        # B. OFFLINE DATASET DUPLICATION VIA ALBUMENTATIONS (Train + Box Only!)\n        enable_aug = getattr(globals(), \"ENABLE_AUGMENTATION\", True)  # True by default optionally\n        if split == \"train\" and len(scaled_boxes) > 0 and enable_aug:\n            # Pass our already 1024x1024 safe arrays\n            img_aug, boxes_aug = apply_dicom_augmentation(img_resized.copy(), scaled_boxes, split)\n            save_image_and_label(img_aug, boxes_aug, f\"{image_id}_aug\", split)\n            num_saved += 1\n\n        elapsed = time.perf_counter() - t0\n        return {\"image_id\": image_id, \"ok\": True, \"reason\": \"done\", \"elapsed\": elapsed, \"num_saved\": num_saved}\n        \n    except Exception as e:\n        return {\"image_id\": image_id, \"ok\": False, \"reason\": str(e), \"elapsed\": time.perf_counter() - t0, \"num_saved\": 0}\n","metadata":{"execution":{"iopub.execute_input":"2026-03-28T01:06:15.174055Z","iopub.status.busy":"2026-03-28T01:06:15.173778Z","iopub.status.idle":"2026-03-28T01:06:15.182024Z","shell.execute_reply":"2026-03-28T01:06:15.181494Z"},"papermill":{"duration":0.015238,"end_time":"2026-03-28T01:06:15.183424+00:00","exception":false,"start_time":"2026-03-28T01:06:15.168186+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"200e747d","cell_type":"markdown","source":"# Chạy export song song + ETA","metadata":{"papermill":{"duration":0.004729,"end_time":"2026-03-28T01:06:15.192628+00:00","exception":false,"start_time":"2026-03-28T01:06:15.187899+00:00","status":"completed"},"tags":[]}},{"id":"4fcba458","cell_type":"code","source":"max_workers = min(8, os.cpu_count() or 4)\nprint(\"max_workers =\", max_workers)\n\ntotal_files = len(dicom_files_selected)\nstart_all = time.perf_counter()\n\ndone = 0\nok_count = 0\ntimes = []\n\nlog_every = 20\n\nwith ThreadPoolExecutor(max_workers=max_workers) as executor:\n    futures = [executor.submit(process_one_dicom, p) for p in dicom_files_selected]\n\n    for future in as_completed(futures):\n        result = future.result()\n        done += 1\n\n        if result[\"ok\"]:\n            ok_count += 1\n            times.append(result[\"elapsed\"])\n\n        if done % log_every == 0 or done == total_files:\n            elapsed_all = time.perf_counter() - start_all\n            avg_per_file = elapsed_all / done\n            speed = done / elapsed_all if elapsed_all > 0 else 0.0\n            remaining = total_files - done\n            eta_sec = remaining * avg_per_file\n\n            avg_worker = np.mean(times) if len(times) > 0 else 0.0\n            p50_worker = np.median(times) if len(times) > 0 else 0.0\n\n            print(\n                f\"[{done}/{total_files}] \"\n                f\"ok={ok_count} | \"\n                f\"wall={elapsed_all:.1f}s | \"\n                f\"speed={speed:.2f} img/s | \"\n                f\"avg_wall/file={avg_per_file:.3f}s | \"\n                f\"avg_worker={avg_worker:.3f}s | \"\n                f\"p50_worker={p50_worker:.3f}s | \"\n                f\"ETA={eta_sec:.1f}s\"\n            )\n\ntotal_elapsed = time.perf_counter() - start_all\nprint(\"\\nDONE\")\nprint(f\"Processed: {done}/{total_files}\")\nprint(f\"Success:   {ok_count}\")\nprint(f\"Total time: {total_elapsed:.2f}s\")\nprint(f\"Overall speed: {done / total_elapsed:.2f} img/s\")","metadata":{"execution":{"iopub.execute_input":"2026-03-28T01:06:15.202605Z","iopub.status.busy":"2026-03-28T01:06:15.202089Z","iopub.status.idle":"2026-03-28T02:55:03.437148Z","shell.execute_reply":"2026-03-28T02:55:03.436092Z"},"papermill":{"duration":6528.242004,"end_time":"2026-03-28T02:55:03.439052+00:00","exception":false,"start_time":"2026-03-28T01:06:15.197048+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"a1ef624a","cell_type":"code","source":"img_ext = \"*.jpg\" if SAVE_AS_JPG else \"*.png\"\n\nprint(\"Train images:\", len(list(IMG_TRAIN_DIR.glob(img_ext))))\nprint(\"Val images:\", len(list(IMG_VAL_DIR.glob(img_ext))))\nprint(\"Train labels:\", len(list(LBL_TRAIN_DIR.glob(\"*.txt\"))))\nprint(\"Val labels:\", len(list(LBL_VAL_DIR.glob(\"*.txt\"))))","metadata":{"execution":{"iopub.execute_input":"2026-03-28T02:55:03.501494Z","iopub.status.busy":"2026-03-28T02:55:03.500985Z","iopub.status.idle":"2026-03-28T02:55:03.602695Z","shell.execute_reply":"2026-03-28T02:55:03.601673Z"},"papermill":{"duration":0.133964,"end_time":"2026-03-28T02:55:03.604216+00:00","exception":false,"start_time":"2026-03-28T02:55:03.470252+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"dadc0fd9","cell_type":"markdown","source":"# Tạo data.yaml","metadata":{"papermill":{"duration":0.030348,"end_time":"2026-03-28T02:55:03.665514+00:00","exception":false,"start_time":"2026-03-28T02:55:03.635166+00:00","status":"completed"},"tags":[]}},{"id":"d53d5400","cell_type":"code","source":"data_yaml = {\n    \"path\": str(OUT_ROOT),\n    \"train\": \"images/train\",\n    \"val\": \"images/val\",\n    \"names\": {i: name for i, name in enumerate(CLASS_NAMES)}\n}\n\nyaml_path = OUT_ROOT / \"data.yaml\"\nwith open(yaml_path, \"w\") as f:\n    yaml.dump(data_yaml, f, sort_keys=False)\n\nprint(\"Saved:\", yaml_path)\nprint(yaml_path.read_text())","metadata":{"execution":{"iopub.execute_input":"2026-03-28T02:55:03.727328Z","iopub.status.busy":"2026-03-28T02:55:03.726918Z","iopub.status.idle":"2026-03-28T02:55:03.734518Z","shell.execute_reply":"2026-03-28T02:55:03.733506Z"},"papermill":{"duration":0.040462,"end_time":"2026-03-28T02:55:03.735958+00:00","exception":false,"start_time":"2026-03-28T02:55:03.695496+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"d25dde0f","cell_type":"code","source":"import matplotlib.pyplot as plt\nfor i in range(10):\n    img_ext = \".jpg\" if SAVE_AS_JPG else \".png\"\n    sample_image_path = list(IMG_TRAIN_DIR.glob(f\"*{img_ext}\"))[i]\n    sample_label_path = LBL_TRAIN_DIR / f\"{sample_image_path.stem}.txt\"\n    \n    img = cv2.imread(str(sample_image_path))\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    h, w = img.shape[:2]\n    \n    if sample_label_path.exists():\n        lines = sample_label_path.read_text().strip().splitlines()\n        for line in lines:\n            parts = line.strip().split()\n            if len(parts) != 5:\n                continue\n    \n            cls_id = int(float(parts[0]))\n            xc, yc, bw, bh = map(float, parts[1:])\n    \n            x1 = int((xc - bw/2) * w)\n            y1 = int((yc - bh/2) * h)\n            x2 = int((xc + bw/2) * w)\n            y2 = int((yc + bh/2) * h)\n    \n            cv2.rectangle(img, (x1, y1), (x2, y2), (255, 0, 0), 2)\n            cv2.putText(\n                img,\n                ID2CLASS[cls_id],\n                (x1, max(20, y1 - 5)),\n                cv2.FONT_HERSHEY_SIMPLEX,\n                0.6,\n                (255, 0, 0),\n                2\n            )\n    \n    plt.figure(figsize=(10, 10))\n    plt.imshow(img)\n    plt.title(sample_image_path.name)\n    plt.axis(\"off\")\n    plt.show()","metadata":{"execution":{"iopub.execute_input":"2026-03-28T02:55:03.79833Z","iopub.status.busy":"2026-03-28T02:55:03.797951Z","iopub.status.idle":"2026-03-28T02:55:07.691935Z","shell.execute_reply":"2026-03-28T02:55:07.691318Z"},"papermill":{"duration":3.933055,"end_time":"2026-03-28T02:55:07.699424+00:00","exception":false,"start_time":"2026-03-28T02:55:03.766369+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"2b577a81","cell_type":"code","source":"img_id = \"053cf0f0a75926ebd53f0265bad6aee4\"\ntmp = df_full[df_full[\"image_id\"] == img_id].copy()\nprint(tmp[[\"image_id\", \"class_name\", \"x_min\", \"y_min\", \"x_max\", \"y_max\"]])\nprint(\"Num rows:\", len(tmp))","metadata":{"execution":{"iopub.execute_input":"2026-03-28T02:55:07.853328Z","iopub.status.busy":"2026-03-28T02:55:07.852596Z","iopub.status.idle":"2026-03-28T02:55:07.873431Z","shell.execute_reply":"2026-03-28T02:55:07.872245Z"},"papermill":{"duration":0.099855,"end_time":"2026-03-28T02:55:07.875252+00:00","exception":false,"start_time":"2026-03-28T02:55:07.775397+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"c0374871","cell_type":"code","source":"img_id = \"0fca086ebe001f784d428aa9973ba691\"\n\nfor p in [LBL_TRAIN_DIR / f\"{img_id}.txt\", LBL_VAL_DIR / f\"{img_id}.txt\"]:\n    if p.exists():\n        print(\"FOUND:\", p)\n        print(p.read_text())","metadata":{"execution":{"iopub.execute_input":"2026-03-28T02:55:08.013087Z","iopub.status.busy":"2026-03-28T02:55:08.01222Z","iopub.status.idle":"2026-03-28T02:55:08.019798Z","shell.execute_reply":"2026-03-28T02:55:08.018845Z"},"papermill":{"duration":0.077403,"end_time":"2026-03-28T02:55:08.02128+00:00","exception":false,"start_time":"2026-03-28T02:55:07.943877+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"ab8a999e","cell_type":"markdown","source":"# Train YOLO","metadata":{"papermill":{"duration":0.070218,"end_time":"2026-03-28T02:55:08.160973+00:00","exception":false,"start_time":"2026-03-28T02:55:08.090755+00:00","status":"completed"},"tags":[]}},{"id":"485d58f8","cell_type":"code","source":"import torch\nfrom ultralytics import YOLO\n\ndevice = 0 if torch.cuda.is_available() else \"cpu\"\nprint(\"Device:\", device)\n\n# model nhỏ để test nhanh\nmodel = YOLO(\"yolov8n.pt\")\n\nmodel.train(\n    fliplr=0.0,             # Explicitly disabled per Research Group (Offline DICOM augs only)\\n\n    hsv_v=0.0,              # Explicitly disabled per Research Group\\n\n    mosaic=0.0,             # Explicitly disabled per Research Group\\n\n    mixup=0.0,              # Explicitly disabled per Research Group\\n\n    data=str(yaml_path),\n    epochs=50,               # test nhanh, có thể đổi 1 / 3 / 10\n    imgsz=IMG_SIZE,\n    batch=16,                # nếu thiếu VRAM thì giảm xuống 4\n    device=device,\n    project=str(OUT_ROOT / \"runs\"),\n    name=\"yolov8n_multiclass_test\",\n    exist_ok=True,\n    optimizer='AdamW',\n    lr0=1e-3,\n    warmup_epochs=3.0,     # ~Warmup lên 1e-4 sau 1000 iters (tuỳ steps/epoch)\n    warmup_bias_lr=1e-4,\n    lrf=0.01,              # Decay xuống 1e-5 (1e-3 * 0.01)\n    cos_lr=True,           # Tương đương thay thế ReduceLROnPlateau cho YOLO\n    patience=6,\n)\n\n# ==========================================\n# OPTIONAL: Checkpoint Averaging (Ensemble)\n# ==========================================\n# Nếu bạn train 3 lần (vd: thay đổi seed/kfold), bạn có thể lấy trung bình params:\n# from ultralytics.models.yolo.detect.val import DetectionValidator\n# ensemble = YOLO([\"runs/train/yolov8n_multiclass_test/weights/best.pt\", \n#                  \"runs/train/yolov8n_multiclass_test2/weights/best.pt\",\n#                  \"runs/train/yolov8n_multiclass_test3/weights/best.pt\"])\n","metadata":{"execution":{"iopub.execute_input":"2026-03-28T02:55:08.299018Z","iopub.status.busy":"2026-03-28T02:55:08.29836Z","iopub.status.idle":"2026-03-28T06:09:01.867061Z","shell.execute_reply":"2026-03-28T06:09:01.866218Z"},"papermill":{"duration":11635.154027,"end_time":"2026-03-28T06:09:03.384266+00:00","exception":false,"start_time":"2026-03-28T02:55:08.230239+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"da983b78","cell_type":"code","source":"","metadata":{"papermill":{"duration":1.384701,"end_time":"2026-03-28T06:09:06.227554+00:00","exception":false,"start_time":"2026-03-28T06:09:04.842853+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}