{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Pipeline NSCLC — Phát hiện nốt phổi trên X-quang và CT\n\n| Giai đoạn | Bài toán | Dataset |\n|---|---|---|\n| A | Phân loại ảnh X-quang (có nốt / không) | NIH ChestX-ray14 |\n| B | Khu trú nốt trên X-quang (bbox) | VinBigData |\n| C | Phát hiện + phân tầng ác tính trên CT | LUNA16 (+ LIDC-IDRI) |\n\nLưu ý: X-quang không phải phương tiện tầm soát NSCLC được khuyến cáo (CT liều thấp mới là),\nnên giai đoạn A/B nên định vị là *sàng lọc cơ hội*, không phải *tầm soát*.","metadata":{}},{"cell_type":"markdown","source":"## Phần 0 — Cài đặt và import","metadata":{}},{"cell_type":"code","source":"!pip install -q SimpleITK pydicom","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:42:49.723824Z","iopub.execute_input":"2026-10-02T15:42:49.724134Z","iopub.status.idle":"2026-10-02T15:42:53.7021Z","shell.execute_reply.started":"2026-10-02T15:42:49.724106Z","shell.execute_reply":"2026-10-02T15:42:53.700858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, sys, json, math, time\nfrom collections import Counter\nfrom pathlib import Path\nfrom types import SimpleNamespace\n\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader, WeightedRandomSampler\n\ncv2.setNumThreads(0)\nplt.rcParams[\"figure.dpi\"] = 110\n\ntry:\n    import SimpleITK as sitk\n    from scipy import ndimage\nexcept ImportError:\n    sitk = ndimage = None\n\nprint(\"SimpleITK:\", getattr(sitk, \"__version__\", \"CHƯA CÀI\"))\nprint(\"torch:\", torch.__version__, \"| CUDA:\", torch.cuda.is_available())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:42:53.704249Z","iopub.execute_input":"2026-10-02T15:42:53.704855Z","iopub.status.idle":"2026-10-02T15:42:57.416954Z","shell.execute_reply.started":"2026-10-02T15:42:53.704818Z","shell.execute_reply":"2026-10-02T15:42:57.415686Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phần 1 — Hằng số và cấu hình\n\n- Trong 14 nhãn NIH, chỉ `Nodule` và `Mass` là dấu hiệu trực tiếp của tổn thương dạng khối.\n- 4 nhãn nhu mô (thâm nhiễm, đông đặc, xẹp phổi, xơ) dễ nhầm với nốt → dùng làm *hard negative*.\n- Cửa sổ HU cho CT phổi: `[-1000, 400]`. Giá trị đệm ngoài biên là −1000 (khí), không dùng 0.","metadata":{}},{"cell_type":"code","source":"NIH_14_LABELS = [\n    \"Atelectasis\", \"Cardiomegaly\", \"Effusion\", \"Infiltration\", \"Mass\", \"Nodule\",\n    \"Pneumonia\", \"Pneumothorax\", \"Consolidation\", \"Edema\", \"Emphysema\",\n    \"Fibrosis\", \"Pleural_Thickening\", \"Hernia\",\n]\nNSCLC_PRIMARY    = [\"Nodule\", \"Mass\"]                                        # positive\nNSCLC_CONFOUNDER = [\"Infiltration\", \"Consolidation\", \"Atelectasis\", \"Fibrosis\"]  # hard negative\nVIN_NSCLC_CLASSES = [\"Nodule/Mass\"]\n\nIMAGENET_MEAN, IMAGENET_STD = [0.485, 0.456, 0.406], [0.229, 0.224, 0.225]\nHU_MIN, HU_MAX, PAD_HU = -1000.0, 400.0, -1000.0\nLUNG_MASK_LABELS = (3, 4)   # 3 = phổi trái, 4 = phổi phải\n\nCFG = SimpleNamespace(\n    # X-quang\n    neg_ratio=2.0, img_size=384, cache_dir=\"/kaggle/working/cache\",\n    # CT LUNA16\n    subsets=[0], max_scans=10, patch_mm=64.0, out_size=64,\n    neg_per_scan=8, min_neg_dist_mm=10.0, use_lung_mask=True,\n    ct_dir=\"/kaggle/working/luna_patches\", lidc_csv=None,\n    # Huấn luyện\n    model=\"convnextv2_base.fcmae_ft_in22k_in1k\", weights_file=None,\n    batch_size=64, epochs=5, lr=1e-4, wd=0.1, layer_decay=0.7, drop_path=0.2,\n    warmup_frac=0.05, max_pos_weight=10.0, workers=2, grad_ckpt=True,\n    max_hours=10.5, out_dir=\"/kaggle/working/run_convnextv2\", seed=42,\n)\nPath(CFG.cache_dir).mkdir(parents=True, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:42:57.418212Z","iopub.execute_input":"2026-10-02T15:42:57.418732Z","iopub.status.idle":"2026-10-02T15:42:57.429087Z","shell.execute_reply.started":"2026-10-02T15:42:57.41868Z","shell.execute_reply":"2026-10-02T15:42:57.427344Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Dò đường dẫn 3 dataset\n\nLoại thư mục `seg-lungs` vì nó chứa mask phổi, không phải ảnh CT.","metadata":{}},{"cell_type":"code","source":"def detect_base_dir() -> Path:\n    if \"KAGGLE_KERNEL_RUN_TYPE\" in os.environ:\n        return Path(\"/kaggle/input\")\n    if \"google.colab\" in sys.modules:\n        return Path(\"/content\")\n    return Path(\"./data\")\n\n\ndef find_dataset_root(base_dir: Path, keywords, max_depth=4, exclude=(\"seg-lungs\",)):\n    \"\"\"BFS tìm thư mục dataset theo từ khoá.\"\"\"\n    if base_dir is None or not base_dir.exists():\n        return None\n    queue = [(base_dir, 0)]\n    while queue:\n        cur, d = queue.pop(0)\n        if d > max_depth:\n            continue\n        try:\n            children = sorted(p for p in cur.iterdir() if p.is_dir())\n        except (PermissionError, FileNotFoundError):\n            continue\n        for ch in children:\n            name = ch.name.lower()\n            if any(ex in name for ex in exclude):\n                continue\n            if any(kw.lower() in name for kw in keywords):\n                return ch\n            queue.append((ch, d + 1))\n    return None\n\n\nbase_dir = detect_base_dir()\nnih_root  = find_dataset_root(base_dir, [\"nih\", \"chestxray8\"])\nvin_root  = find_dataset_root(base_dir, [\"vinbig\"])\nluna_root = find_dataset_root(base_dir, [\"luna\"])\nfor name, p in [(\"NIH\", nih_root), (\"VinBigData\", vin_root), (\"LUNA16\", luna_root)]:\n    print(f\"{name:12s}: {p}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:42:57.430454Z","iopub.execute_input":"2026-10-02T15:42:57.430881Z","iopub.status.idle":"2026-10-02T15:42:57.468368Z","shell.execute_reply.started":"2026-10-02T15:42:57.430837Z","shell.execute_reply":"2026-10-02T15:42:57.467177Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phần 2 — NIH ChestX-ray14\n\nBa điểm phải xử lý:\n1. Cột tuổi có giá trị phi lý (>100) → đặt `NaN`, không xoá dòng.\n2. **Chia split theo bệnh nhân** — 112k ảnh chỉ từ 30k bệnh nhân, chia theo ảnh sẽ rò rỉ dữ liệu.\n3. Nhãn được khai thác tự động từ báo cáo bằng NLP → nhãn yếu, phải nêu trong báo cáo.","metadata":{}},{"cell_type":"code","source":"def nih_data_dir(root: Path) -> Path:\n    return root / \"data\" if (root / \"data\").exists() else root\n\n\ndef nih_split_by_patient(data_dir: Path, df: pd.DataFrame, val_frac=0.15, seed=42) -> pd.Series:\n    \"\"\"Ưu tiên split chính thức; tách val từ train_val THEO patient_id.\"\"\"\n    rng = np.random.default_rng(seed)\n    split = pd.Series(\"train\", index=df.index, dtype=object)\n    tv_file, te_file = data_dir / \"train_val_list.txt\", data_dir / \"test_list.txt\"\n\n    if tv_file.exists() and te_file.exists():\n        te = set(te_file.read_text().split())\n        split[df[\"image_id\"].isin(te)] = \"test\"\n        tv_patients = df.loc[~df[\"image_id\"].isin(te), \"patient_id\"].unique()\n        print(\"[NIH] Dùng split chính thức của NIH.\")\n    else:\n        pts = df[\"patient_id\"].unique().copy(); rng.shuffle(pts)\n        test_pts = set(pts[:int(0.20 * len(pts))])\n        split[df[\"patient_id\"].isin(test_pts)] = \"test\"\n        tv_patients = np.array([p for p in pts if p not in test_pts])\n        print(\"[NIH] Không có file split -> chia ngẫu nhiên theo patient_id.\")\n\n    tv_patients = np.array(sorted(tv_patients)); rng.shuffle(tv_patients)\n    val_pts = set(tv_patients[:int(val_frac * len(tv_patients))])\n    split[(split != \"test\") & df[\"patient_id\"].isin(val_pts)] = \"val\"\n\n    if (df.assign(_s=split).groupby(\"patient_id\")[\"_s\"].nunique() > 1).any():\n        raise RuntimeError(\"Rò rỉ dữ liệu: có bệnh nhân nằm ở nhiều split.\")\n    return split\n\n\ndef build_nih_index(root: Path) -> pd.DataFrame:\n    data_dir = nih_data_dir(root)\n    df = pd.read_csv(data_dir / \"Data_Entry_2017.csv\")\n    df = df.loc[:, ~df.columns.str.startswith(\"Unnamed\")].rename(columns={\n        \"Image Index\": \"image_id\", \"Finding Labels\": \"findings\",\n        \"Patient ID\": \"patient_id\", \"Patient Age\": \"age\",\n        \"Patient Gender\": \"gender\", \"View Position\": \"view\"})\n\n    # tuổi phi lý -> NaN (chỉ dùng tuổi để ghép cặp ca đối chứng)\n    df[\"age\"] = pd.to_numeric(df[\"age\"], errors=\"coerce\")\n    df.loc[(df[\"age\"] < 1) | (df[\"age\"] > 100), \"age\"] = np.nan\n\n    labels = df[\"findings\"].str.split(\"|\")\n    for lab in NIH_14_LABELS:\n        df[lab] = labels.apply(lambda xs, l=lab: int(l in xs))\n    df[\"no_finding\"]  = (df[\"findings\"] == \"No Finding\").astype(int)\n    df[\"nodule_mass\"] = df[NSCLC_PRIMARY].max(axis=1)\n    df[\"confounder\"]  = df[NSCLC_CONFOUNDER].max(axis=1)\n    df[\"nm_group\"] = np.select(\n        [df[\"nodule_mass\"] == 1, df[\"no_finding\"] == 1, df[\"confounder\"] == 1],\n        [\"positive\", \"clean_negative\", \"hard_negative\"], default=\"other_abnormal\")\n\n    df[\"split\"] = nih_split_by_patient(data_dir, df)\n    print(f\"[NIH] {len(df)} ảnh / {df['patient_id'].nunique()} bệnh nhân | \"\n          f\"Nodule hoặc Mass: {int(df['nodule_mass'].sum())}\")\n    return df\n\n\ndef build_nih_path_map(root: Path) -> dict:\n    \"\"\"image_id -> đường dẫn file PNG (quét 1 lần).\"\"\"\n    data_dir = nih_data_dir(root)\n    mapping = {}\n    for folder in sorted(p for p in data_dir.iterdir() if p.is_dir() and \"images_\" in p.name.lower()):\n        sub = folder / \"images\" if (folder / \"images\").is_dir() else folder\n        for p in sub.iterdir():\n            if p.suffix.lower() == \".png\":\n                mapping[p.name] = str(p)\n    print(f\"[NIH] Đã lập chỉ mục {len(mapping)} file ảnh.\")\n    return mapping\n\n\ndf_nih = build_nih_index(nih_root)\npath_map = build_nih_path_map(nih_root)\ndf_nih.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:42:57.471313Z","iopub.execute_input":"2026-10-02T15:42:57.471815Z","iopub.status.idle":"2026-10-02T15:43:00.558733Z","shell.execute_reply.started":"2026-10-02T15:42:57.471784Z","shell.execute_reply":"2026-10-02T15:43:00.557811Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Biểu đồ thống kê NIH","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(2, 2, figsize=(13, 8))\n\n# 1. Phân bố 14 nhãn\ncnt = df_nih[NIH_14_LABELS].sum().sort_values()\ncolors = [\"#d62728\" if l in NSCLC_PRIMARY else\n          \"#ff7f0e\" if l in NSCLC_CONFOUNDER else \"#7f7f7f\" for l in cnt.index]\nax[0, 0].barh(cnt.index, cnt.values, color=colors)\nax[0, 0].set_title(\"Số ảnh theo nhãn (đỏ = Nodule/Mass, cam = hard negative)\")\nfor i, v in enumerate(cnt.values):\n    ax[0, 0].text(v, i, f\" {v}\", va=\"center\", fontsize=7)\n\n# 2. Nhóm NSCLC\ng = df_nih[\"nm_group\"].value_counts()\nax[0, 1].bar(g.index, g.values, color=\"#1f77b4\")\nax[0, 1].set_title(\"Phân bố nhóm NSCLC\"); ax[0, 1].tick_params(axis=\"x\", rotation=15)\nfor i, v in enumerate(g.values):\n    ax[0, 1].text(i, v, str(v), ha=\"center\", va=\"bottom\", fontsize=8)\n\n# 3. Split\ns = df_nih[\"split\"].value_counts()\nax[1, 0].bar(s.index, s.values, color=\"#2ca02c\")\nax[1, 0].set_title(\"Phân bố split (chia theo bệnh nhân)\")\nfor i, v in enumerate(s.values):\n    ax[1, 0].text(i, v, str(v), ha=\"center\", va=\"bottom\", fontsize=8)\n\n# 4. Tư thế chụp x nhãn Nodule/Mass  -> kiểm tra nguy cơ học tắt theo tư thế\nct = pd.crosstab(df_nih[\"view\"], df_nih[\"nodule_mass\"], normalize=\"index\")\nct.plot(kind=\"bar\", stacked=True, ax=ax[1, 1], color=[\"#cccccc\", \"#d62728\"])\nax[1, 1].set_title(\"Tỷ lệ Nodule/Mass theo tư thế chụp\"); ax[1, 1].tick_params(axis=\"x\", rotation=0)\nax[1, 1].legend([\"không\", \"có\"], fontsize=8)\n\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:43:00.559954Z","iopub.execute_input":"2026-10-02T15:43:00.560319Z","iopub.status.idle":"2026-10-02T15:43:01.500882Z","shell.execute_reply.started":"2026-10-02T15:43:00.560281Z","shell.execute_reply":"2026-10-02T15:43:01.499623Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Ảnh mẫu NIH","metadata":{}},{"cell_type":"code","source":"def show_nih_samples(df, path_map, n=4):\n    pos = df[df[\"nm_group\"] == \"positive\"].head(n)\n    neg = df[df[\"nm_group\"] == \"clean_negative\"].head(n)\n    fig, axes = plt.subplots(2, n, figsize=(3.2 * n, 7))\n    for row, (part, tag) in enumerate([(pos, \"Nodule/Mass\"), (neg, \"No Finding\")]):\n        for k, (_, r) in enumerate(part.iterrows()):\n            img = cv2.imread(path_map[r[\"image_id\"]], cv2.IMREAD_GRAYSCALE)\n            axes[row, k].imshow(img, cmap=\"gray\"); axes[row, k].axis(\"off\")\n            axes[row, k].set_title(f\"{tag}\\n{r['findings'][:28]}\", fontsize=8)\n    plt.tight_layout(); plt.show()\n\nshow_nih_samples(df_nih, path_map)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:43:01.502248Z","iopub.execute_input":"2026-10-02T15:43:01.502846Z","iopub.status.idle":"2026-10-02T15:43:04.066005Z","shell.execute_reply.started":"2026-10-02T15:43:01.5028Z","shell.execute_reply":"2026-10-02T15:43:04.064187Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Lọc subset hướng NSCLC\n\nNếu chỉ lấy `Nodule/Mass` làm positive và `No Finding` làm negative thì bài toán trở nên\ndễ giả tạo: mô hình chỉ cần phân biệt \"phổi có bất thường\" với \"phổi sạch\". Vì vậy subset\ngiữ cả **hard negative** (tổn thương nhu mô dễ nhầm), và ghép cặp ca đối chứng theo\ngiới tính × tư thế chụp × nhóm tuổi để mô hình không học tắt theo tư thế AP/PA.","metadata":{}},{"cell_type":"code","source":"def make_nsclc_subset(df, neg_ratio=2.0, hard_frac=0.5, seed=42):\n    rng = np.random.default_rng(seed)\n    strata_cols = [\"gender\", \"view\", \"age_bin\"]\n    out = []\n    for _, g in df.groupby(\"split\"):\n        g = g.copy()\n        g[\"age_bin\"] = pd.cut(g[\"age\"], bins=[0, 30, 40, 50, 60, 70, 80, 200], labels=False)\n        pos = g[g[\"nm_group\"] == \"positive\"]\n        if pos.empty:\n            continue\n        pos_strata = Counter(map(tuple, g.loc[g[\"nm_group\"] == \"positive\", strata_cols]\n                                 .fillna(-1).astype(object).values))\n        total_w = sum(pos_strata.values()) or 1\n        picks = [pos]\n        n_hard = int(len(pos) * neg_ratio * hard_frac)\n        for group, n_total in [(\"hard_negative\", n_hard),\n                               (\"clean_negative\", int(len(pos) * neg_ratio) - n_hard)]:\n            pool = g[g[\"nm_group\"] == group].copy()\n            if pool.empty or n_total <= 0:\n                continue\n            pool[\"_st\"] = list(map(tuple, pool[strata_cols].fillna(-1).astype(object).values))\n            chosen = []\n            for stratum, w in pos_strata.items():          # lấy theo đúng tỷ lệ tầng\n                quota = int(round(n_total * w / total_w))\n                cand = pool.index[pool[\"_st\"] == stratum].to_numpy()\n                if quota and len(cand):\n                    chosen += rng.choice(cand, min(quota, len(cand)), replace=False).tolist()\n            rest = pool.index.difference(pd.Index(chosen)).to_numpy()  # bù tầng rỗng\n            if n_total - len(chosen) > 0 and len(rest):\n                chosen += rng.choice(rest, min(n_total - len(chosen), len(rest)),\n                                     replace=False).tolist()\n            picks.append(g.loc[chosen])\n        out.append(pd.concat(picks))\n\n    sub = pd.concat(out).drop(columns=[\"age_bin\", \"_st\"], errors=\"ignore\")\n    return sub.sample(frac=1.0, random_state=seed).reset_index(drop=True)\n\n\ndf_use = make_nsclc_subset(df_nih, neg_ratio=CFG.neg_ratio)\nprint(f\"Subset: {len(df_use)} / {len(df_nih)} ảnh gốc\")\nct = pd.crosstab(df_use[\"split\"], df_use[\"nm_group\"])\nprint(ct.to_string())\n\nct.plot(kind=\"bar\", stacked=True, figsize=(7, 4),\n        color=[\"#9edae5\", \"#ff7f0e\", \"#7f7f7f\", \"#d62728\"])\nplt.title(\"Subset NSCLC: thành phần từng split\"); plt.ylabel(\"số ảnh\")\nplt.xticks(rotation=0); plt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:43:04.067961Z","iopub.execute_input":"2026-10-02T15:43:04.068419Z","iopub.status.idle":"2026-10-02T15:43:05.039848Z","shell.execute_reply.started":"2026-10-02T15:43:04.068345Z","shell.execute_reply":"2026-10-02T15:43:05.038977Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phần 3 — VinBigData: bbox `Nodule/Mass`\n\nMỗi ảnh được **3 bác sĩ đọc độc lập** → cùng một tổn thương xuất hiện tới 3 hộp lệch nhau.\nPhải gộp cụm theo IoU rồi lấy trung bình trước khi đưa vào detector, và giữ cột `n_raters`\n(số bác sĩ đồng thuận). `min_raters=2` cho nhãn sạch hơn nhưng ít mẫu hơn.\n\nĐọc DICOM đúng thứ tự: **Modality LUT → VOI LUT → đảo nếu `MONOCHROME1`**.","metadata":{}},{"cell_type":"code","source":"def read_dicom_xray(path) -> np.ndarray:\n    import pydicom\n    from pydicom.pixel_data_handlers.util import apply_modality_lut, apply_voi_lut\n    ds = pydicom.dcmread(str(path))\n    img = ds.pixel_array.astype(np.float32)\n    try:\n        img = apply_voi_lut(apply_modality_lut(img, ds), ds)\n    except Exception:\n        pass\n    if getattr(ds, \"PhotometricInterpretation\", \"\") == \"MONOCHROME1\":\n        img = img.max() - img\n    lo, hi = np.percentile(img, [0.5, 99.5])       # chống pixel cháy sáng\n    img = np.clip(img, lo, hi)\n    return (img - img.min()) / (img.max() - img.min() + 1e-8)\n\n\ndef fuse_boxes_iou(boxes: np.ndarray, iou_thr=0.4) -> np.ndarray:\n    \"\"\"(N,4) -> (M,5): gộp cụm bbox trùng nhau, cột cuối = số bác sĩ đồng thuận.\"\"\"\n    if len(boxes) == 0:\n        return boxes\n    area = (boxes[:, 2] - boxes[:, 0]) * (boxes[:, 3] - boxes[:, 1]) # S = (x_max - x_min) * (y_max - y_min)\n    boxes = boxes[np.argsort(-area)]\n    used, fused = np.zeros(len(boxes), bool), []\n    for i in range(len(boxes)):\n        if used[i]:\n            continue\n        cluster, used[i] = [boxes[i]], True\n        for j in range(i + 1, len(boxes)):\n            if used[j]:\n                continue\n            xx1, yy1 = max(boxes[i, 0], boxes[j, 0]), max(boxes[i, 1], boxes[j, 1])\n            xx2, yy2 = min(boxes[i, 2], boxes[j, 2]), min(boxes[i, 3], boxes[j, 3])\n            inter = max(0, xx2 - xx1) * max(0, yy2 - yy1)\n            a1 = (boxes[i, 2] - boxes[i, 0]) * (boxes[i, 3] - boxes[i, 1])\n            a2 = (boxes[j, 2] - boxes[j, 0]) * (boxes[j, 3] - boxes[j, 1])\n            if inter / (a1 + a2 - inter + 1e-8) >= iou_thr:\n                cluster.append(boxes[j]); used[j] = True\n        fused.append(np.mean(cluster, axis=0).tolist() + [len(cluster)])\n    return np.array(fused)\n\n\ndef build_vinbig_index(vin_root: Path, min_raters=1):\n    raw = pd.read_csv(vin_root / \"train.csv\")\n    nod = raw[raw[\"class_name\"].isin(VIN_NSCLC_CLASSES)].dropna(subset=[\"x_min\"])\n    rows = []\n    for img_id, g in nod.groupby(\"image_id\"):\n        boxes = g[[\"x_min\", \"y_min\", \"x_max\", \"y_max\"]].to_numpy(np.float32)\n        for x1, y1, x2, y2, n in fuse_boxes_iou(boxes):\n            if n >= min_raters:\n                rows.append(dict(image_id=img_id, x_min=x1, y_min=y1, x_max=x2, y_max=y2,\n                                 n_raters=int(n)))\n    out = pd.DataFrame(rows)\n    print(f\"[Vin] {len(raw)} annotation / {raw['image_id'].nunique()} ảnh\")\n    print(f\"[Vin] Nodule/Mass: {len(nod)} bbox thô -> {len(out)} sau hợp nhất \"\n          f\"(min_raters={min_raters}), trên {out['image_id'].nunique()} ảnh\")\n    return raw, out\n\n\nvin_raw, vin_idx = build_vinbig_index(vin_root, min_raters=2)\nvin_idx.to_csv(f\"{CFG.cache_dir}/vinbig_nodule_mass.csv\", index=False)\nvin_idx.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:43:05.041037Z","iopub.execute_input":"2026-10-02T15:43:05.041887Z","iopub.status.idle":"2026-10-02T15:43:05.721214Z","shell.execute_reply.started":"2026-10-02T15:43:05.041843Z","shell.execute_reply":"2026-10-02T15:43:05.720392Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Biểu đồ thống kê VinBigData","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 3, figsize=(15, 4))\n\n# 1. Phân bố 15 lớp tổn thương\ncnt = vin_raw[\"class_name\"].value_counts().sort_values()\ncolors = [\"#d62728\" if c in VIN_NSCLC_CLASSES else \"#7f7f7f\" for c in cnt.index]\nax[0].barh(cnt.index, cnt.values, color=colors)\nax[0].set_xscale(\"log\"); ax[0].set_title(\"Số annotation theo lớp (log)\")\nax[0].tick_params(axis=\"y\", labelsize=7)\n\n# 2. Số bác sĩ đồng thuận trên mỗi bbox đã hợp nhất\nr = vin_idx[\"n_raters\"].value_counts().sort_index()\nax[1].bar(r.index.astype(str), r.values, color=\"#1f77b4\")\nax[1].set_title(\"Số bác sĩ đồng thuận / bbox\"); ax[1].set_xlabel(\"n_raters\")\nfor i, v in enumerate(r.values):\n    ax[1].text(i, v, str(v), ha=\"center\", va=\"bottom\", fontsize=8)\n\n# 3. Kích thước bbox\nw = vin_idx[\"x_max\"] - vin_idx[\"x_min\"]\nh = vin_idx[\"y_max\"] - vin_idx[\"y_min\"]\nax[2].scatter(w, h, s=8, alpha=0.4, color=\"#d62728\")\nax[2].set_xlabel(\"rộng (px)\"); ax[2].set_ylabel(\"cao (px)\")\nax[2].set_title(f\"Kích thước bbox (trung vị {w.median():.0f}x{h.median():.0f} px)\")\n\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:43:05.722294Z","iopub.execute_input":"2026-10-02T15:43:05.722978Z","iopub.status.idle":"2026-10-02T15:43:06.422981Z","shell.execute_reply.started":"2026-10-02T15:43:05.72294Z","shell.execute_reply":"2026-10-02T15:43:06.422156Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Ảnh mẫu VinBigData kèm bbox\n\nBbox trong `train.csv` nằm trong hệ toạ độ của DICOM gốc. Nếu dùng bản PNG đã resize thì phải nhân tỷ lệ — code dưới đọc `train_meta.csv` (nếu có) để tự quy đổi.","metadata":{}},{"cell_type":"code","source":"def find_vin_image(vin_root: Path, image_id: str):\n    for ext in (\".dicom\", \".dcm\", \".png\", \".jpg\", \".jpeg\"):\n        p = vin_root / \"train\" / f\"{image_id}{ext}\"\n        if p.exists():\n            return p\n    hits = [p for p in vin_root.rglob(f\"{image_id}.*\") if p.suffix.lower() != \".csv\"]\n    return hits[0] if hits else None\n\n\ndef load_vin_image(path):\n    if path.suffix.lower() in (\".dicom\", \".dcm\"):\n        return read_dicom_xray(path)\n    img = cv2.imread(str(path), cv2.IMREAD_GRAYSCALE)\n    return img.astype(np.float32) / 255.0\n\n\n# kích thước DICOM gốc (nếu dataset là bản PNG resize thì cần để quy đổi bbox)\nmeta_path = vin_root / \"train_meta.csv\"\nvin_meta = pd.read_csv(meta_path).set_index(\"image_id\") if meta_path.exists() else None\nprint(\"train_meta.csv:\", \"có\" if vin_meta is not None else \"không có (giả định ảnh gốc)\")\n\n\ndef show_vin_samples(n=4):\n    ids = (vin_idx.groupby(\"image_id\").size().sort_values(ascending=False).index[:n])\n    fig, axes = plt.subplots(1, n, figsize=(4.2 * n, 4.6))\n    axes = np.atleast_1d(axes)\n    for ax, img_id in zip(axes, ids):\n        p = find_vin_image(vin_root, img_id)\n        if p is None:\n            ax.text(0.5, 0.5, \"không tìm thấy file ảnh\", ha=\"center\"); ax.axis(\"off\"); continue\n        img = load_vin_image(p)\n        sy = sx = 1.0\n        if vin_meta is not None and img_id in vin_meta.index:      # quy đổi nếu ảnh đã resize\n            sy = img.shape[0] / float(vin_meta.loc[img_id, \"dim0\"])\n            sx = img.shape[1] / float(vin_meta.loc[img_id, \"dim1\"])\n        ax.imshow(img, cmap=\"gray\")\n        for _, b in vin_idx[vin_idx[\"image_id\"] == img_id].iterrows():\n            x1, y1 = b.x_min * sx, b.y_min * sy\n            ax.add_patch(plt.Rectangle((x1, y1), (b.x_max - b.x_min) * sx,\n                                       (b.y_max - b.y_min) * sy,\n                                       fill=False, color=\"red\", lw=1.8))\n            ax.text(x1, y1 - 8, f\"{b.n_raters} BS\", color=\"red\", fontsize=8)\n        ax.set_title(f\"{img_id[:12]}...\", fontsize=8); ax.axis(\"off\")\n    plt.suptitle(\"VinBigData — Nodule/Mass (hộp đỏ = bbox sau hợp nhất)\", fontsize=11)\n    plt.tight_layout(); plt.show()\n\n\nshow_vin_samples()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:43:06.424115Z","iopub.execute_input":"2026-10-02T15:43:06.424523Z","iopub.status.idle":"2026-10-02T15:43:21.756062Z","shell.execute_reply.started":"2026-10-02T15:43:06.424494Z","shell.execute_reply":"2026-10-02T15:43:21.754869Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### So sánh bbox thô và bbox sau hợp nhất trên cùng một ảnh","metadata":{}},{"cell_type":"code","source":"img_id = vin_idx.groupby(\"image_id\").size().sort_values(ascending=False).index[0]\np = find_vin_image(vin_root, img_id)\nimg = load_vin_image(p)\nsy = sx = 1.0\nif vin_meta is not None and img_id in vin_meta.index:\n    sy = img.shape[0] / float(vin_meta.loc[img_id, \"dim0\"])\n    sx = img.shape[1] / float(vin_meta.loc[img_id, \"dim1\"])\n\nraw_b = vin_raw[(vin_raw[\"image_id\"] == img_id) &\n                (vin_raw[\"class_name\"].isin(VIN_NSCLC_CLASSES))].dropna(subset=[\"x_min\"])\n\nfig, axes = plt.subplots(1, 2, figsize=(11, 5.5))\nfor ax, (boxes, title, color) in zip(axes, [\n        (raw_b[[\"x_min\", \"y_min\", \"x_max\", \"y_max\"]].to_numpy(), \"Bbox thô (3 bác sĩ)\", \"orange\"),\n        (vin_idx.loc[vin_idx[\"image_id\"] == img_id,\n                     [\"x_min\", \"y_min\", \"x_max\", \"y_max\"]].to_numpy(), \"Sau hợp nhất IoU\", \"red\")]):\n    ax.imshow(img, cmap=\"gray\"); ax.axis(\"off\"); ax.set_title(f\"{title} — {len(boxes)} hộp\")\n    for x1, y1, x2, y2 in boxes:\n        ax.add_patch(plt.Rectangle((x1 * sx, y1 * sy), (x2 - x1) * sx, (y2 - y1) * sy,\n                                   fill=False, color=color, lw=1.8))\nplt.tight_layout(); \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:43:21.75794Z","iopub.execute_input":"2026-10-02T15:43:21.758342Z","iopub.status.idle":"2026-10-02T15:43:25.790331Z","shell.execute_reply.started":"2026-10-02T15:43:21.758301Z","shell.execute_reply":"2026-10-02T15:43:25.789162Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phần 4 — Hợp nhất Dữ liệu (Harmonization) và Lưu Cache\n\nĐể tối ưu hóa luồng dữ liệu (Data Pipeline Optimization) và tránh nút thắt cổ chai I/O khi huấn luyện, chúng ta sẽ gộp chung **33.621 ảnh NIH (PNG)** và **15.000 ảnh VinBigData (DICOM)** vào một khối dữ liệu duy nhất.\n\n**Các bước thực hiện trong khâu này:**\n1. **Đồng bộ nhãn bệnh lý:** Ánh xạ các nhãn của VinBigData sang hệ 14 nhãn chuẩn của NIH.\n2. **Chuẩn hóa Bounding Box:** Đọc siêu tốc kích thước ảnh gốc từ DICOM header để ép tọa độ Bbox về dải `[0, 1]`. Đối với ảnh NIH (không có Bbox), tọa độ được gán mặc định là `[0, 0, 0, 0]`.\n3. **Lưu Cache siêu tốc (.npy):** Đọc, thu nhỏ ảnh (dùng nội suy `INTER_AREA` để bảo toàn chi tiết các nốt phổi nhỏ) và nén toàn bộ 48.621 ảnh thành một file Memory-Mapped `.npy`. Lát nữa khi Train, mô hình sẽ tải trực tiếp từ đây để GPU chạy 100% công suất mà không cần chờ CPU giải mã file gốc.","metadata":{}},{"cell_type":"code","source":"# Bảng ánh xạ nhãn VinBigData sang định dạng 14 nhãn của NIH\nimport pydicom  # THIẾU Ở BẢN GỐC -> pydicom.dcmread() bên dưới NameError, bị try/except nuốt,\n                 # kết quả là MỌI bbox VinBig lặng lẽ giữ nguyên giá trị khởi tạo [0,0,0,0]\n\nvin_label_dict = {\n    \"Atelectasis\": [\"Atelectasis\"],\n    \"Cardiomegaly\": [\"Cardiomegaly\"],\n    \"Pleural effusion\": [\"Effusion\"],\n    \"Infiltration\": [\"Infiltration\"],\n    \"Nodule/Mass\": [\"Nodule\", \"Mass\", \"nodule_mass\"], \n    \"Consolidation\": [\"Consolidation\"],\n    \"Pleural thickening\": [\"Pleural_Thickening\"],\n    \"Pulmonary fibrosis\": [\"Fibrosis\"],\n    \"Pneumothorax\": [\"Pneumothorax\"]\n}\n\nprint(\"Đang trích xuất nhãn từ VinBigData...\")\n# Lấy nhãn từ vin_raw (đã đọc ở Phần 3)\nvin_grouped = vin_raw.groupby('image_id')['class_name'].apply(set).reset_index()\n\n# Khởi tạo ma trận zero cho tất cả các nhãn và bbox\nfor col in NIH_14_LABELS + [\"nodule_mass\", \"no_finding\", \"confounder\", \"x_min\", \"y_min\", \"x_max\", \"y_max\"]:\n    vin_grouped[col] = 0.0\n    \n# Gán nhãn 1 (Positive)\nfor idx, row in vin_grouped.iterrows():\n    classes = row['class_name']\n    if \"No finding\" in classes:\n        vin_grouped.at[idx, \"no_finding\"] = 1\n    for v_class in classes:\n        if v_class in vin_label_dict:\n            for n_class in vin_label_dict[v_class]:\n                vin_grouped.at[idx, n_class] = 1\n\n# Tính nhãn phụ và phân split\nvin_grouped[\"confounder\"] = vin_grouped[NSCLC_CONFOUNDER].max(axis=1)\nvin_grouped[\"nm_group\"] = np.where(vin_grouped[\"nodule_mass\"] == 1, \"positive\", \"other_abnormal\")\nvin_grouped[\"patient_id\"] = \"VIN_\" + vin_grouped[\"image_id\"]\n\nnp.random.seed(42) # Cố định seed để chia ổn định qua các lần chạy\nvin_patients = vin_grouped[\"patient_id\"].unique()\nnp.random.shuffle(vin_patients)\n\nn_pts = len(vin_patients)\n# Lấy index để cắt mảng: 80% train, 10% val, 10% test\ntrain_end = int(0.8 * n_pts)\nval_end = int(0.9 * n_pts)\n\nval_pts = vin_patients[train_end:val_end]\ntest_pts = vin_patients[val_end:]\n\n# Gán nhãn split tương ứng\nvin_grouped[\"split\"] = \"train\" # Mặc định 80% đầu tiên là train\nvin_grouped.loc[vin_grouped[\"patient_id\"].isin(val_pts), \"split\"] = \"val\"\nvin_grouped.loc[vin_grouped[\"patient_id\"].isin(test_pts), \"split\"] = \"test\"\n\n# Đọc danh sách path ảnh thật\nvin_train_dir = vin_root / \"train\"\nvin_paths = list(vin_train_dir.glob(\"*.dicom\")) + list(vin_train_dir.glob(\"*.dcm\"))\nvin_path_map = {p.stem: str(p) for p in vin_paths}\n\nprint(\"Đang đọc kích thước ảnh DICOM và chuẩn hóa tọa độ Bounding Box...\")\n# Mỗi ảnh lấy box LỚN NHẤT (sau hợp nhất ở Phần 3) làm target — KHÔNG lấy hộp bao trùm\n# min/max của mọi tổn thương: ảnh có 2 nốt cách xa nhau (vd trái/phải phổi) mà lấy bao\n# trùm sẽ ra một box khổng lồ phủ cả vùng giữa, không còn ý nghĩa định vị.\nvin_idx_area = vin_idx.assign(_area=(vin_idx.x_max - vin_idx.x_min) * (vin_idx.y_max - vin_idx.y_min))\nbbox_dict = (vin_idx_area.sort_values(\"_area\", ascending=False)\n             .drop_duplicates(\"image_id\", keep=\"first\")\n             .set_index(\"image_id\")[[\"x_min\", \"y_min\", \"x_max\", \"y_max\"]].to_dict(\"index\"))\n\n# Trích xuất kích thước và chuẩn hóa bbox\nfor idx, row in vin_grouped.iterrows():\n    img_id = row['image_id']\n    if img_id in bbox_dict and img_id in vin_path_map:\n        try:\n            # Đọc DICOM Header (Cực nhanh nhờ bỏ qua mảng pixels bằng stop_before_pixels=True)\n            ds = pydicom.dcmread(vin_path_map[img_id], stop_before_pixels=True)\n            w, h = float(ds.Columns), float(ds.Rows)\n            \n            box = bbox_dict[img_id]\n            vin_grouped.at[idx, 'x_min'] = box['x_min'] / w\n            vin_grouped.at[idx, 'y_min'] = box['y_min'] / h\n            vin_grouped.at[idx, 'x_max'] = box['x_max'] / w\n            vin_grouped.at[idx, 'y_max'] = box['y_max'] / h\n        except Exception:\n            pass # Bỏ qua nếu file ảnh lỗi\n\n# Lọc các ảnh thực sự tồn tại trên ổ đĩa\nvin_grouped = vin_grouped[vin_grouped[\"image_id\"].isin(vin_path_map.keys())].reset_index(drop=True)\n\nn_box = int(((vin_grouped[\"x_max\"] > vin_grouped[\"x_min\"]) & (vin_grouped[\"y_max\"] > vin_grouped[\"y_min\"])).sum())\nprint(f\"[Vin] Chuẩn hoá bbox thành công cho {n_box} / {len(vin_grouped)} ảnh \"\n      f\"({'OK — import pydicom đã có' if n_box > 0 else 'VẪN LÀ 0 -> còn lỗi khác, kiểm tra lại!'})\")\n\n# Khởi tạo giá trị giả 0.0 cho tọa độ Bbox của bảng NIH (vì bộ NIH không có nhãn Bbox)\nfor col in ['x_min', 'y_min', 'x_max', 'y_max']:\n    if col not in df_use.columns:\n        df_use[col] = 0.0\n\n# Cuối cùng, gộp DataFrame của NIH (df_use) và VinBigData (vin_grouped) lại\ndf_combined = pd.concat([df_use, vin_grouped], ignore_index=True)\npath_map_combined = {**path_map, **vin_path_map}\n\nprint(f\"Tổng ảnh NIH: {len(df_use)} | VinBigData: {len(vin_grouped)} | Tổng Dataset Hợp Nhất: {len(df_combined)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:43:25.792054Z","iopub.execute_input":"2026-10-02T15:43:25.792485Z","iopub.status.idle":"2026-10-02T15:43:35.764426Z","shell.execute_reply.started":"2026-10-02T15:43:25.792441Z","shell.execute_reply":"2026-10-02T15:43:35.763301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cấu hình lại hàm tiền xử lý để đọc lẫn lộn cả PNG (NIH) và DICOM (VinBigData)\ndef preprocess_mixed(path, out_size=384):\n    p_str = str(path).lower()\n    if p_str.endswith('.dicom') or p_str.endswith('.dcm'):\n        try:\n            img_float = read_dicom_xray(Path(path))\n            img = (img_float * 255).astype(np.uint8)\n        except Exception: return None\n    else:\n        img = cv2.imread(str(path), cv2.IMREAD_GRAYSCALE)\n        \n    if img is None: return None\n    interp = cv2.INTER_AREA if img.shape[0] > out_size else cv2.INTER_CUBIC\n    return cv2.resize(img, (out_size, out_size), interpolation=interp)\n\n\ndef _worker_mixed(args): \n    return preprocess_mixed(*args)\n\n# Chạy cache lại ra một bộ npy hỗn hợp mới\nfrom concurrent.futures import ProcessPoolExecutor\nout_dir = Path(CFG.cache_dir)\nnpy_path_mixed = out_dir / f\"cxr_mixed_{CFG.img_size}.npy\"\nidx_path_mixed = out_dir / f\"cxr_mixed_{CFG.img_size}_index.csv\"\n\ndf_cache = df_combined.copy()\ndf_cache[\"path\"] = df_cache[\"image_id\"].map(path_map_combined)\ndf_cache = df_cache[df_cache[\"path\"].notna()].reset_index(drop=True)\n\narr = np.lib.format.open_memmap(npy_path_mixed, mode=\"w+\", dtype=np.uint8, shape=(len(df_cache), CFG.img_size, CFG.img_size))\nok = np.zeros(len(df_cache), bool)\n\nprint(\"Đang tạo mảng numpy hỗn hợp (NIH + VinBigData)...\")\nwith ProcessPoolExecutor(max_workers=os.cpu_count() or 2) as ex:\n    for i, img in enumerate(ex.map(_worker_mixed, ((p, CFG.img_size) for p in df_cache[\"path\"]), chunksize=64)):\n        if img is not None: arr[i], ok[i] = img, True\n        if (i+1) % 2500 == 0: print(f\"  ... {i+1}/{len(df_cache)}\")\n    arr.flush()\n\ndf_cache[\"cache_row\"], df_cache[\"cache_ok\"] = np.arange(len(df_cache)), ok\ndf_cache.to_csv(idx_path_mixed, index=False)\nprint(f\"[Xong] File chỉ mục: {idx_path_mixed}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T15:43:35.768591Z","iopub.execute_input":"2026-10-02T15:43:35.76914Z","iopub.status.idle":"2026-10-02T17:43:24.237978Z","shell.execute_reply.started":"2026-10-02T15:43:35.769108Z","shell.execute_reply":"2026-10-02T17:43:24.234229Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Kiểm tra cache bằng mắt\n\nẢnh đen thui hoặc đảo âm bản là lỗi phát hiện trong 10 giây ở đây nhưng tốn hàng giờ nếu chỉ thấy sau khi train.","metadata":{}},{"cell_type":"code","source":"arr_mixed = np.load(npy_path_mixed, mmap_mode=\"r\")\nidx_df_mixed = pd.read_csv(idx_path_mixed)\n\nfig = plt.figure(figsize=(14, 7))\nfor k, group in enumerate([\"positive\", \"hard_negative\", \"clean_negative\"]):\n    rows = idx_df_mixed.index[idx_df_mixed[\"nm_group\"] == group][:4]\n    for j, i in enumerate(rows):\n        ax = fig.add_subplot(3, 5, k*5 + j + 1)\n        ax.imshow(arr_mixed[int(idx_df_mixed.loc[i, \"cache_row\"])], cmap=\"gray\")\n        ax.axis(\"off\")\n        title_text = str(idx_df_mixed.loc[i, 'findings'])[:22] if 'findings' in idx_df_mixed.columns else \"VinBigData\"\n        ax.set_title(f\"{group}\\n{title_text}\", fontsize=7)\n\nax = fig.add_subplot(3, 5, 5)\nax.hist(np.asarray(arr_mixed[:200]).ravel(), bins=50, color=\"#1f77b4\")\nax.set_title(\"Histogram pixel (0-255)\", fontsize=8)\nax.set_yticks([])\nplt.tight_layout()\nplt.show()\n\nprint(\"Dải pixel:\", arr_mixed[:50].min(), \"-\", arr_mixed[:50].max())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:43:24.248625Z","iopub.execute_input":"2026-10-02T17:43:24.24933Z","iopub.status.idle":"2026-10-02T17:43:26.800696Z","shell.execute_reply.started":"2026-10-02T17:43:24.249255Z","shell.execute_reply":"2026-10-02T17:43:26.799481Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phần 5 — Dataset và DataLoader\n\n**Không lật ngang**: lật ngang đảo phải/trái (tim, thuỳ phổi) và làm sai lệch nhãn phụ thuộc bên.\nCác phép còn lại nhẹ và bảo toàn tổn thương: crop 0,85–1,0, xoay ±7°, sáng/tương phản ±15%.","metadata":{}},{"cell_type":"code","source":"def build_transforms(img_size, train):\n    import albumentations as A\n    from albumentations.pytorch import ToTensorV2\n    if not train:\n        return A.Compose([A.Resize(img_size, img_size),\n                          A.Normalize(IMAGENET_MEAN, IMAGENET_STD), ToTensorV2()])\n    v2 = int(A.__version__.split(\".\")[0]) >= 2     # API đổi giữa 1.4.x và 2.x\n    rrc = (A.RandomResizedCrop(size=(img_size, img_size), scale=(0.85, 1.0)) if v2\n           else A.RandomResizedCrop(height=img_size, width=img_size, scale=(0.85, 1.0)))\n    return A.Compose([\n        rrc,\n        A.Affine(rotate=(-7, 7), translate_percent=(-0.03, 0.03), scale=(0.95, 1.05), p=0.7),\n        A.RandomBrightnessContrast(0.15, 0.15, p=0.7),\n        A.Normalize(IMAGENET_MEAN, IMAGENET_STD), ToTensorV2(),\n    ])\n\n\nclass CXRDataset:\n    def __init__(self, index_df, npy_path, label_cols, img_size=384, train=False):\n        self.df = index_df.reset_index(drop=True)\n        self.arr = np.load(str(npy_path), mmap_mode=\"r\")\n        self.rows = self.df[\"cache_row\"].to_numpy()\n        self.labels = self.df[label_cols].to_numpy(np.float32)\n        self.tf = build_transforms(img_size, train)\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, i):\n        gray = np.asarray(self.arr[self.rows[i]])\n        img = np.repeat(gray[:, :, None], 3, axis=2)   # 1ch -> 3ch cho backbone ImageNet\n        return self.tf(image=img)[\"image\"], self.labels[i]\n\n\ndef make_loaders(index_csv, npy_path, label_cols, img_size=384, batch_size=32,\n                 workers=2, balance_col=None):\n    import torch\n    from torch.utils.data import DataLoader, WeightedRandomSampler\n    df = pd.read_csv(index_csv)\n    df = df[df[\"cache_ok\"]] if \"cache_ok\" in df else df\n    loaders = {}\n    for split in [\"train\", \"val\", \"test\"]:\n        part = df[df[\"split\"] == split]\n        if part.empty:\n            continue\n        is_train = split == \"train\"\n        ds = CXRDataset(part, npy_path, label_cols, img_size, train=is_train)\n        sampler, shuffle = None, is_train\n        if is_train and balance_col:\n            y = part[balance_col].to_numpy().astype(int)\n            w = (1.0 / np.maximum(np.bincount(y, minlength=2), 1))[y]\n            sampler = WeightedRandomSampler(torch.as_tensor(w, dtype=torch.double), len(w), True)\n            shuffle = False\n        loaders[split] = DataLoader(ds, batch_size=batch_size, shuffle=shuffle,\n                                    sampler=sampler, num_workers=workers,\n                                    pin_memory=True, drop_last=is_train)\n        print(f\"[loader] {split}: {len(ds)} ảnh / {len(loaders[split])} batch\")\n    return loaders","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:43:26.802445Z","iopub.execute_input":"2026-10-02T17:43:26.802887Z","iopub.status.idle":"2026-10-02T17:43:26.822772Z","shell.execute_reply.started":"2026-10-02T17:43:26.802854Z","shell.execute_reply":"2026-10-02T17:43:26.821318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Xem augmentation có làm hỏng ảnh không","metadata":{}},{"cell_type":"code","source":"# Lấy 1 mẫu từ tập hỗn hợp để test augment\nsample = idx_df_mixed[idx_df_mixed[\"split\"] == \"train\"].head(1)\nds_aug = CXRDataset(sample, npy_path_mixed, NIH_14_LABELS, CFG.img_size, train=True)\n\nfig, axes = plt.subplots(1, 5, figsize=(14, 3.2))\n# Lấy ảnh gốc từ arr_mixed\naxes[0].imshow(arr_mixed[int(sample.iloc[0][\"cache_row\"])], cmap=\"gray\")\naxes[0].set_title(\"gốc\", fontsize=9); axes[0].axis(\"off\")\n\nfor k in range(1, 5):\n    x, _ = ds_aug[0]\n    # Đảo chuẩn hóa (De-normalize) để hiển thị\n    img = x.numpy().transpose(1, 2, 0) * np.array(IMAGENET_STD) + np.array(IMAGENET_MEAN)\n    axes[k].imshow(np.clip(img, 0, 1))\n    axes[k].set_title(f\"augment {k}\", fontsize=9)\n    axes[k].axis(\"off\")\n    \nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:43:26.824281Z","iopub.execute_input":"2026-10-02T17:43:26.824773Z","iopub.status.idle":"2026-10-02T17:43:31.004003Z","shell.execute_reply.started":"2026-10-02T17:43:26.824741Z","shell.execute_reply":"2026-10-02T17:43:31.002404Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phần 6 — CT LUNA16: cắt patch 3D\n\n3 nguyên tắc:\n1. Chỉ resample **trong patch**: cắt khối vật lý 64 mm³ rồi resize về 64³ → luôn đúng 1,0 mm/voxel.\n2. Đổi toạ độ thế giới → voxel bằng SimpleITK, không tự tính `(world − origin)/spacing`\n   (sai ở ca có direction cosine âm).\n3. Negative **không phải** \"mọi vị trí ngoài `annotations.csv`\" — các vị trí trong\n   `annotations_excluded.csv` là tổn thương thật, phải loại khỏi pool negative.","metadata":{}},{"cell_type":"code","source":"def subset_of(path: Path) -> int:\n    for part in path.parts:\n        if part.lower().startswith(\"subset\") and part[6:].isdigit():\n            return int(part[6:])\n    return -1\n\n\ndef list_ct_scans(luna_root: Path, subsets=None) -> dict:\n    scans = {p.stem: p for p in luna_root.rglob(\"*.mhd\")\n             if \"seg-lungs\" not in str(p).lower()\n             and (subsets is None or subset_of(p) in subsets)}\n    print(f\"[LUNA] {len(scans)} ca CT (subsets={subsets})\")\n    return scans\n\n\ndef load_luna_csv(luna_root: Path, name: str):\n    hits = list(luna_root.rglob(name))\n    if not hits:\n        print(f\"[LUNA] Không tìm thấy {name}\")\n        return None\n    df = pd.read_csv(hits[0])\n    print(f\"[LUNA] {name}: {len(df)} dòng\")\n    return df\n\n\ndef read_scan(mhd_path: Path):\n    itk = sitk.ReadImage(str(mhd_path))\n    vol = sitk.GetArrayFromImage(itk).astype(np.float32)      # (z, y, x) HU\n    return itk, vol, np.array(itk.GetSpacing()[::-1], float)\n\n\ndef world_to_voxel(itk, world_xyz):\n    idx = itk.TransformPhysicalPointToContinuousIndex([float(v) for v in world_xyz])\n    return np.array(idx[::-1], float)                          # -> (z, y, x)\n\n\ndef extract_patch(vol, center_zyx, spacing_zyx, patch_mm=64.0, out_size=64):\n    \"\"\"Cắt khối patch_mm^3 quanh tâm rồi resample về out_size^3 -> uint8 theo cửa sổ HU.\"\"\"\n    size_vox = np.maximum(np.ceil(patch_mm / np.asarray(spacing_zyx)).astype(int), 2)\n    start = np.round(np.asarray(center_zyx) - size_vox / 2).astype(int)\n    end, shape = start + size_vox, np.array(vol.shape)\n    pad_lo, pad_hi = np.maximum(0, -start), np.maximum(0, end - shape)\n    s, e = np.clip(start, 0, shape), np.clip(end, 0, shape)\n    crop = vol[s[0]:e[0], s[1]:e[1], s[2]:e[2]]\n    if crop.size == 0:\n        return np.zeros((out_size,) * 3, np.uint8)\n    if pad_lo.any() or pad_hi.any():                           # đệm bằng khí, không phải 0\n        crop = np.pad(crop, tuple(zip(pad_lo.tolist(), pad_hi.tolist())),\n                      constant_values=PAD_HU)\n    zoom = [out_size / crop.shape[i] for i in range(3)]\n    if not np.allclose(zoom, 1.0):\n        crop = ndimage.zoom(crop, zoom, order=1, mode=\"nearest\")\n    crop = crop[:out_size, :out_size, :out_size]\n    crop = np.pad(crop, [(0, out_size - crop.shape[i]) for i in range(3)], mode=\"edge\")\n    return ((np.clip(crop, HU_MIN, HU_MAX) - HU_MIN) / (HU_MAX - HU_MIN) * 255).astype(np.uint8)\n\n\ndef uint8_to_hu(patch):\n    return patch.astype(np.float32) / 255 * (HU_MAX - HU_MIN) + HU_MIN","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:43:31.005588Z","iopub.execute_input":"2026-10-02T17:43:31.005934Z","iopub.status.idle":"2026-10-02T17:43:31.024753Z","shell.execute_reply.started":"2026-10-02T17:43:31.005902Z","shell.execute_reply":"2026-10-02T17:43:31.023456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def world_to_voxel_batch(itk, world_xyz_arr):\n    \"\"\"Vector hoá world_to_voxel cho N điểm cùng lúc bằng ma trận affine (direction,\n    spacing, origin) của chính ảnh ITK — khớp itk.TransformPhysicalPointToContinuousIndex\n    tới sai số ~1e-13 (đã kiểm chứng kể cả với direction cosine âm), nhanh hơn vòng lặp\n    Python gọi SimpleITK từng điểm một khi cần lọc hàng trăm/nghìn candidate mỗi ca CT.\"\"\"\n    D = np.array(itk.GetDirection()).reshape(3, 3)\n    spacing = np.array(itk.GetSpacing())\n    origin = np.array(itk.GetOrigin())\n    idx_xyz = (np.linalg.inv(D) @ (world_xyz_arr - origin).T).T / spacing\n    return idx_xyz[:, ::-1]  # xyz -> zyx\n\n\ndef sample_scan_patches(itk, vol, spacing, nodules, candidates, lung_mask,\n                        patch_mm, out_size, neg_per_scan, min_neg_dist_mm, rng, excluded):\n    patches, metas = [], []\n    nod_w = nodules[[\"coordX\", \"coordY\", \"coordZ\"]].to_numpy(float)\n    avoid = nod_w\n    if excluded is not None and len(excluded):\n        exc = excluded[[\"coordX\", \"coordY\", \"coordZ\"]].to_numpy(float)\n        avoid = np.vstack([nod_w, exc]) if len(nod_w) else exc\n\n    for _, r in nodules.iterrows():                             # positive\n        c = world_to_voxel(itk, (r.coordX, r.coordY, r.coordZ))\n        patches.append(extract_patch(vol, c, spacing, patch_mm, out_size))\n        metas.append(dict(seriesuid=r.seriesuid, label=1, diameter_mm=float(r.diameter_mm),\n                          coordX=r.coordX, coordY=r.coordY, coordZ=r.coordZ))\n\n    if candidates is None or neg_per_scan <= 0:\n        return patches, metas\n    neg = candidates[candidates[\"class\"] == 0]\n    if neg.empty:\n        return patches, metas\n\n    cand_w = neg[[\"coordX\", \"coordY\", \"coordZ\"]].to_numpy(float)\n    keep = np.ones(len(cand_w), bool)\n    if len(avoid):                                              # cách mọi tổn thương đã biết\n        keep &= np.linalg.norm(cand_w[:, None] - avoid[None], axis=2).min(1) >= min_neg_dist_mm\n    if lung_mask is not None:                                   # ưu tiên nằm trong phổi (vector hoá)\n        v_all = np.round(world_to_voxel_batch(itk, cand_w)).astype(int)\n        shape = np.array(lung_mask.shape)\n        in_bounds = np.all(v_all >= 0, axis=1) & np.all(v_all < shape, axis=1)\n        in_lung = np.zeros(len(cand_w), bool)\n        ib = np.flatnonzero(in_bounds)\n        vv = v_all[ib]\n        in_lung[ib] = lung_mask[vv[:, 0], vv[:, 1], vv[:, 2]]\n        keep &= in_lung\n\n    pool = np.flatnonzero(keep)\n    if len(pool) == 0:\n        pool = np.arange(len(cand_w))\n    for _, r in neg.iloc[rng.choice(pool, min(neg_per_scan, len(pool)), False)].iterrows():\n        c = world_to_voxel(itk, (r.coordX, r.coordY, r.coordZ))\n        patches.append(extract_patch(vol, c, spacing, patch_mm, out_size))\n        metas.append(dict(seriesuid=r.seriesuid, label=0, diameter_mm=np.nan,\n                          coordX=r.coordX, coordY=r.coordY, coordZ=r.coordZ))\n    return patches, metas\n\n\ndef build_patch_dataset(luna_root: Path, out_dir: Path, subsets=None, patch_mm=64.0,\n                        out_size=64, neg_per_scan=8, min_neg_dist_mm=10.0,\n                        use_lung_mask=True, shard_size=500, max_scans=None, seed=42):\n    rng = np.random.default_rng(seed)\n    out_dir.mkdir(parents=True, exist_ok=True)\n\n    ann = load_luna_csv(luna_root, \"annotations.csv\")\n    cand = load_luna_csv(luna_root, \"candidates_V2.csv\")\n    if cand is None:\n        cand = load_luna_csv(luna_root, \"candidates.csv\")\n    exc = load_luna_csv(luna_root, \"annotations_excluded.csv\")\n    scans = list_ct_scans(luna_root, subsets)\n    uids = sorted(scans)[:max_scans] if max_scans else sorted(scans)\n\n    by = lambda d: dict(tuple(d.groupby(\"seriesuid\"))) if d is not None else {}\n    ann_by, cand_by, exc_by = by(ann), by(cand), by(exc)\n    empty = pd.DataFrame(columns=[\"seriesuid\", \"coordX\", \"coordY\", \"coordZ\", \"diameter_mm\"])\n\n    buf_p, buf_m, shard_id, index_rows = [], [], 0, []\n    for k, uid in enumerate(uids, 1):\n        try:\n            itk, vol, spacing = read_scan(scans[uid])\n        except Exception as e:\n            print(f\"  [skip] {uid[:20]}: {e}\"); continue\n\n        mask = None\n        if use_lung_mask:\n            mp = next((p for p in luna_root.rglob(f\"{uid}.mhd\") if \"seg-lungs\" in str(p).lower()), None)\n            if mp is not None:\n                mask = np.isin(sitk.GetArrayFromImage(sitk.ReadImage(str(mp))), LUNG_MASK_LABELS)\n\n        ps, ms = sample_scan_patches(itk, vol, spacing, ann_by.get(uid, empty),\n                                     cand_by.get(uid), mask, patch_mm, out_size,\n                                     neg_per_scan, min_neg_dist_mm, rng, exc_by.get(uid))\n        for m in ms:\n            m[\"subset\"] = subset_of(scans[uid])\n        buf_p += ps; buf_m += ms\n        del vol, mask\n\n        while len(buf_p) >= shard_size:\n            arr = np.stack(buf_p[:shard_size]).astype(np.uint8)\n            name = f\"patches_{shard_id:04d}.npz\"\n            np.savez_compressed(out_dir / name, patches=arr)\n            index_rows += [{**m, \"shard\": name, \"row\": i} for i, m in enumerate(buf_m[:shard_size])]\n            buf_p, buf_m, shard_id = buf_p[shard_size:], buf_m[shard_size:], shard_id + 1\n        print(f\"  [{k}/{len(uids)}] patch: {len(index_rows) + len(buf_p)}\")\n\n    if buf_p:\n        arr = np.stack(buf_p).astype(np.uint8)\n        name = f\"patches_{shard_id:04d}.npz\"\n        np.savez_compressed(out_dir / name, patches=arr)\n        index_rows += [{**m, \"shard\": name, \"row\": i} for i, m in enumerate(buf_m)]\n\n    idx = pd.DataFrame(index_rows)\n    # LUNA16 quy định 10-fold theo subset0..9; nếu không suy được thì chia theo CA BỆNH\n    # (tuyệt đối không chia theo patch — nhiều patch cùng một ca sẽ gây rò rỉ dữ liệu)\n    if (idx[\"subset\"] >= 0).all():\n        idx[\"fold\"] = idx[\"subset\"]\n    else:\n        uids_ = sorted(idx[\"seriesuid\"].unique())\n        rng.shuffle(uids_)\n        idx[\"fold\"] = idx[\"seriesuid\"].map({u: i % 5 for i, u in enumerate(uids_)})\n    idx_path = out_dir / \"patch_index.csv\"\n    idx.to_csv(idx_path, index=False)\n    print(f\"\\n[LUNA] {len(idx)} patch | positive {int((idx.label==1).sum())} | \"\n          f\"negative {int((idx.label==0).sum())} | {idx.seriesuid.nunique()} ca CT\")\n    return idx_path\n\n\nct_idx_path = build_patch_dataset(\n    luna_root, Path(CFG.ct_dir), subsets=CFG.subsets, patch_mm=CFG.patch_mm,\n    out_size=CFG.out_size, neg_per_scan=CFG.neg_per_scan,\n    min_neg_dist_mm=CFG.min_neg_dist_mm, use_lung_mask=CFG.use_lung_mask,\n    max_scans=CFG.max_scans, seed=CFG.seed)\nct_idx = pd.read_csv(ct_idx_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:43:31.026572Z","iopub.execute_input":"2026-10-02T17:43:31.026995Z","iopub.status.idle":"2026-10-02T17:44:07.527817Z","shell.execute_reply.started":"2026-10-02T17:43:31.026949Z","shell.execute_reply":"2026-10-02T17:44:07.526566Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Ảnh mẫu LUNA16\n\nNốt phải nằm **ở giữa** patch. Lệch ra rìa hoặc patch đen thui = khâu đổi toạ độ đang sai.","metadata":{}},{"cell_type":"code","source":"def show_ct_patches(ct_idx, n=4):\n    def load(r):\n        return np.load(Path(CFG.ct_dir) / r[\"shard\"])[\"patches\"][int(r[\"row\"])]\n\n    pos = ct_idx[ct_idx.label == 1].head(n)\n    neg = ct_idx[ct_idx.label == 0].head(n)\n    fig, axes = plt.subplots(3, n, figsize=(3.2 * n, 9.5))\n    c = CFG.out_size // 2\n    for k, (_, r) in enumerate(pos.iterrows()):\n        hu = uint8_to_hu(load(r))\n        for row, (sl, name) in enumerate([(hu[c], \"axial\"), (hu[:, c], \"coronal\")]):\n            ax = axes[row, k]\n            ax.imshow(sl, cmap=\"gray\", vmin=HU_MIN, vmax=HU_MAX)\n            ax.axhline(c, color=\"r\", lw=0.4); ax.axvline(c, color=\"r\", lw=0.4); ax.axis(\"off\")\n            ax.set_title(f\"POS {name} · d={r['diameter_mm']:.1f}mm\", fontsize=8)\n    for k, (_, r) in enumerate(neg.iterrows()):\n        ax = axes[2, k]\n        ax.imshow(uint8_to_hu(load(r))[c], cmap=\"gray\", vmin=HU_MIN, vmax=HU_MAX)\n        ax.axis(\"off\"); ax.set_title(\"NEG axial\", fontsize=8)\n    plt.tight_layout(); plt.show()\n\n\nshow_ct_patches(ct_idx)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:44:07.530022Z","iopub.execute_input":"2026-10-02T17:44:07.530603Z","iopub.status.idle":"2026-10-02T17:44:10.136739Z","shell.execute_reply.started":"2026-10-02T17:44:07.530568Z","shell.execute_reply":"2026-10-02T17:44:10.135505Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Biểu đồ thống kê LUNA16","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 3, figsize=(15, 4))\n\nv = ct_idx[\"label\"].value_counts().sort_index()\nax[0].bar([\"negative\", \"positive\"], [v.get(0, 0), v.get(1, 0)], color=[\"#7f7f7f\", \"#d62728\"])\nax[0].set_title(\"Số patch theo nhãn\")\nfor i, val in enumerate([v.get(0, 0), v.get(1, 0)]):\n    ax[0].text(i, val, str(val), ha=\"center\", va=\"bottom\", fontsize=9)\n\nd = ct_idx.loc[ct_idx.label == 1, \"diameter_mm\"].dropna()\nax[1].hist(d, bins=20, color=\"#d62728\")\nax[1].set_title(f\"Đường kính nốt (trung vị {d.median():.1f} mm)\"); ax[1].set_xlabel(\"mm\")\n\nper_scan = ct_idx.groupby(\"seriesuid\")[\"label\"].sum()\nax[2].bar(range(len(per_scan)), per_scan.values, color=\"#1f77b4\")\nax[2].set_title(\"Số nốt positive mỗi ca CT\"); ax[2].set_xlabel(\"ca CT\")\n\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:44:10.138159Z","iopub.execute_input":"2026-10-02T17:44:10.138716Z","iopub.status.idle":"2026-10-02T17:44:10.65364Z","shell.execute_reply.started":"2026-10-02T17:44:10.13868Z","shell.execute_reply":"2026-10-02T17:44:10.652514Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phần 7 — Nhãn ác tính (LIDC-IDRI)\n\n\nĐây là **đánh giá chủ quan của bác sĩ trên hình ảnh**, không phải chẩn đoán giải phẫu bệnh —\nbáo cáo không được gọi đầu ra là \"chẩn đoán ung thư\".","metadata":{}},{"cell_type":"code","source":"def attach_malignancy_from_csv(patch_index_csv, seg_csv_path, match_mm=5.0):\n    \"\"\"Ghép nhãn ác tính vào LUNA16 dựa trên cột mal_bool từ annotations_for_segmentation.csv.\"\"\"\n    \n    # 1. Đọc index của LUNA16 (đã tạo ở Phần 6)\n    idx = pd.read_csv(patch_index_csv)\n    idx[\"mal_binary\"] = -1  # Mặc định -1 cho những nốt không ghép được\n    idx[\"mal_match_dist_mm\"] = np.nan\n    \n    # 2. Đọc file CSV chứa mal_bool\n    print(f\"Đọc dữ liệu ác tính từ: {seg_csv_path}\")\n    df_mal = pd.read_csv(seg_csv_path)\n    \n    # Nhóm dữ liệu theo seriesuid để tìm kiếm nhanh hơn\n    mal_by_uid = dict(tuple(df_mal.groupby(\"seriesuid\")))\n    \n    # 3. Lặp qua các nốt có nhãn tổn thương (label == 1) trong LUNA16\n    for i in idx.index[idx[\"label\"] == 1]:\n        uid = idx.at[i, \"seriesuid\"]\n        if uid not in mal_by_uid: \n            continue\n            \n        sub = mal_by_uid[uid]\n        \n        # Lấy toạ độ không gian của nốt LUNA16\n        p_target = idx.loc[i, [\"coordX\", \"coordY\", \"coordZ\"]].to_numpy(float)\n        \n        # Tính khoảng cách Euclidean 3D đến tất cả các nốt trong file CSV của uid này\n        p_sub = sub[[\"coordX\", \"coordY\", \"coordZ\"]].to_numpy(float)\n        dist = np.linalg.norm(p_sub - p_target, axis=1)\n        \n        # Lấy nốt gần nhất\n        j = int(np.argmin(dist))\n        \n        # Nếu khoảng cách <= ngưỡng cho phép (match_mm), tiến hành ghép nhãn\n        if dist[j] <= match_mm:\n            # Lấy trực tiếp giá trị boolean True/False từ cột mal_bool\n            is_malignant = sub.iloc[j][\"mal_bool\"]\n            \n            # Gán: True -> 1 (Ác tính), False -> 0 (Lành tính)\n            idx.at[i, \"mal_binary\"] = 1 if is_malignant else 0\n            idx.at[i, \"mal_match_dist_mm\"] = float(dist[j])\n            \n    # Thống kê kết quả\n    n_pos = int((idx.label == 1).sum())\n    n_ok = int((idx.mal_binary != -1).sum())\n    print(f\"[malignancy] Ghép thành công {n_ok} / {n_pos} nốt ({100*n_ok/max(n_pos, 1):.1f}%)\")\n    if n_ok > 0:\n        d = idx.loc[idx[\"mal_match_dist_mm\"].notna(), \"mal_match_dist_mm\"]\n        print(f\"[malignancy] Khoảng cách ghép: trung vị {d.median():.2f} mm, lớn nhất {d.max():.2f} mm \"\n              f\"— nếu các số này lớn bất thường (gần bằng match_mm) thì khả năng hệ toạ độ của \"\n              f\"{Path(seg_csv_path).name} không khớp LUNA16, cần kiểm tra lại trước khi tin kết quả.\")\n    \n    # Ghi đè ra file mới\n    out_path = Path(patch_index_csv).with_name(\"patch_index_malignancy.csv\")\n    idx.to_csv(out_path, index=False)\n    \n    # Vẽ biểu đồ phân bố nếu có dữ liệu\n    if n_ok > 0:\n        mal_counts = idx.loc[idx.label == 1, \"mal_binary\"].map({0: \"Benign (False)\", 1: \"Malignant (True)\"}).value_counts()\n        mal_counts.plot(kind=\"bar\", color=[\"#1f77b4\", \"#d62728\"], figsize=(6, 4))\n        plt.title(\"Phân bố nhãn Ác tính (mal_bool)\")\n        plt.ylabel(\"Số lượng nốt\")\n        plt.xticks(rotation=0)\n        plt.tight_layout()\n        plt.show()\n        \n    return idx\n\n# THỰC THI\n# Đường dẫn cũ bị hardcode theo dạng URL Kaggle (/kaggle/input/datasets/<user>/<slug>/...),\n# khác quy ước mount thực tế lúc chạy kernel (/kaggle/input/<slug>/...) — dò động bằng\n# rglob cho nhất quán với cách mọi dataset khác trong notebook này được tìm (find_dataset_root).\n_seg_hits = list(Path(\"/kaggle/input\").rglob(\"annotations_for_segmentation.csv\"))\nif _seg_hits:\n    SEG_CSV_PATH = _seg_hits[0]\n    print(f\"[path] Tìm thấy CSV ác tính: {SEG_CSV_PATH}\")\n    ct_idx_mal = attach_malignancy_from_csv(ct_idx_path, SEG_CSV_PATH)\nelse:\n    print(\"[path] Không tìm thấy annotations_for_segmentation.csv trong /kaggle/input — \"\n          \"hãy add dataset 'joyou159/annotations-for-segmentation' vào notebook rồi chạy lại cell này.\")\n    ct_idx_mal = pd.read_csv(ct_idx_path); ct_idx_mal[\"mal_binary\"] = -1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:44:10.655062Z","iopub.execute_input":"2026-10-02T17:44:10.6556Z","iopub.status.idle":"2026-10-02T17:46:17.65652Z","shell.execute_reply.started":"2026-10-02T17:44:10.655554Z","shell.execute_reply":"2026-10-02T17:46:17.655264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hiển thị 10 nốt dương tính kèm nhãn ác tính (1: Ác tính, 0: Lành tính, -1: Không tìm thấy)\ndisplay(ct_idx_mal[ct_idx_mal['label'] == 1][['seriesuid', 'coordX', 'coordY', 'coordZ', 'diameter_mm', 'mal_binary']].head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:46:17.658101Z","iopub.execute_input":"2026-10-02T17:46:17.658682Z","iopub.status.idle":"2026-10-02T17:46:17.674496Z","shell.execute_reply.started":"2026-10-02T17:46:17.658638Z","shell.execute_reply":"2026-10-02T17:46:17.673499Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phần 8 — Dataset 3D cho patch CT\n\nPatch lưu ở 64³ nhưng train ở 48³, phần dư dùng để random crop (jitter vị trí).\nLật cả 3 trục và xoay 90° trong mặt phẳng axial đều hợp lệ vì nốt phổi không có hướng ưu tiên.","metadata":{}},{"cell_type":"code","source":"class LunaPatchDataset:\n    def __init__(self, index_df, shard_dir, crop_size=48, train=False, target_col=\"label\"):\n        self.df = index_df.reset_index(drop=True)\n        self.shard_dir, self.crop = Path(shard_dir), crop_size\n        self.train, self.target_col = train, target_col\n        self._cache, self.rng = {}, np.random.default_rng(0)\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, i):\n        import torch\n        r = self.df.iloc[i]\n        if r[\"shard\"] not in self._cache:\n            self._cache[r[\"shard\"]] = np.load(self.shard_dir / r[\"shard\"])[\"patches\"]\n        patch = self._cache[r[\"shard\"]][int(r[\"row\"])].astype(np.float32) / 255.0\n\n        s = patch.shape[0]\n        off = ([int(self.rng.integers(0, s - self.crop + 1)) for _ in range(3)]\n               if self.train else [(s - self.crop) // 2] * 3)\n        patch = patch[off[0]:off[0]+self.crop, off[1]:off[1]+self.crop, off[2]:off[2]+self.crop]\n\n        if self.train:\n            for ax in range(3):\n                if self.rng.random() < 0.5:\n                    patch = np.flip(patch, ax)\n            k = int(self.rng.integers(0, 4))\n            if k:\n                patch = np.rot90(patch, k, axes=(1, 2))\n            patch = np.clip(patch + self.rng.normal(0, 0.02, patch.shape), 0, 1).astype(np.float32)\n\n        x = torch.from_numpy(np.ascontiguousarray(patch))[None]      # (1, D, H, W)\n        return x, torch.tensor(float(r[self.target_col]), dtype=torch.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:46:17.676166Z","iopub.execute_input":"2026-10-02T17:46:17.677286Z","iopub.status.idle":"2026-10-02T17:46:17.701953Z","shell.execute_reply.started":"2026-10-02T17:46:17.677247Z","shell.execute_reply":"2026-10-02T17:46:17.700926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phần 9 — Huấn luyện ConvNeXt V2 trên X-quang\n\n- **Layer-wise LR decay 0,7**: tầng nông (cạnh, kết cấu) học chậm, tầng sâu học nhanh.\n- **fp16**: GPU P100 của Kaggle không hỗ trợ bf16.\n- **Gradient checkpointing**: giảm ~40% VRAM, chậm ~25% — cần khi chạy Base @384 trên GPU 16 GB.\n- **`pos_weight` chặn trên ở 10**: lớp cực hiếm (Hernia) sẽ làm loss mất ổn định nếu để tự do.\n- Ngoài AUROC còn tính **Average Precision** — với dữ liệu mất cân bằng mạnh, AP phản ánh\n  chất lượng thực tế tốt hơn.","metadata":{}},{"cell_type":"code","source":"def build_model(num_classes, model_name, weights_file, grad_ckpt, drop_path=0.2):\n    import timm\n    kw = dict(num_classes=num_classes, drop_path_rate=drop_path, pretrained=True)\n    if weights_file:                         # Kaggle offline\n        kw[\"pretrained_cfg_overlay\"] = dict(file=weights_file)\n    model = timm.create_model(model_name, **kw)\n    if grad_ckpt:\n        model.set_grad_checkpointing(True)\n    print(f\"[model] {model_name} | {sum(p.numel() for p in model.parameters())/1e6:.1f}M tham số\")\n    return model\n\n\ndef build_optimizer(model, lr, wd, layer_decay):\n    if layer_decay and layer_decay < 1.0:\n        try:\n            from timm.optim import param_groups_layer_decay\n            groups = param_groups_layer_decay(model, weight_decay=wd, layer_decay=layer_decay)\n            for g in groups:                 # param_groups_layer_decay() chỉ gắn lr_scale,\n                g[\"lr\"] = g[\"lr_scale\"] * lr  # KHÔNG tự nhân vào lr — thiếu dòng này thì mọi\n                                              # nhóm tham số train cùng một lr, layer decay vô tác dụng\n            print(f\"[optim] layer decay {layer_decay} ({len(groups)} nhóm, \"\n                  f\"lr {min(g['lr'] for g in groups):.2e} -> {max(g['lr'] for g in groups):.2e})\")\n            return torch.optim.AdamW(groups, lr=lr)\n        except Exception as e:\n            print(f\"[optim] không dùng được layer decay ({e}) -> tách head/backbone\")\n    head = [p for n, p in model.named_parameters() if n.startswith(\"head.\")]\n    body = [p for n, p in model.named_parameters() if not n.startswith(\"head.\")]\n    return torch.optim.AdamW([{\"params\": body, \"lr\": lr * 0.1},\n                              {\"params\": head, \"lr\": lr}], weight_decay=wd)\n\n\ndef cosine_with_warmup(opt, warmup, total, min_ratio=0.02):\n    def fn(step):\n        if step < warmup:\n            return (step + 1) / max(1, warmup)\n        p = (step - warmup) / max(1, total - warmup)\n        return min_ratio + (1 - min_ratio) * 0.5 * (1 + math.cos(math.pi * min(p, 1.0)))\n    return torch.optim.lr_scheduler.LambdaLR(opt, fn)\n\n\ndef evaluate(model, loader, device, label_cols, split_name=\"val\"):\n    from sklearn.metrics import average_precision_score, roc_auc_score\n    model.eval()\n    L, Y = [], []\n    with torch.no_grad():\n        for x, y in loader:\n            x = x.to(device, non_blocking=True, memory_format=torch.channels_last)\n            with torch.autocast(\"cuda\", dtype=torch.float16, enabled=device.type == \"cuda\"):\n                out = model(x)\n            L.append(out.float().cpu()); Y.append(y)\n    logits, y = torch.cat(L).numpy(), torch.cat(Y).numpy()\n    if y.ndim == 1:\n        y = y[:, None]\n\n    rows = []\n    for j, name in enumerate(label_cols):\n        n_pos = int(y[:, j].sum())\n        auc = ap = float(\"nan\")\n        if 0 < n_pos < len(y):\n            auc = roc_auc_score(y[:, j], logits[:, j])\n            ap = average_precision_score(y[:, j], logits[:, j])\n        rows.append(dict(label=name, n_pos=n_pos, auroc=auc, ap=ap))\n    res = pd.DataFrame(rows)\n    macro = res[\"auroc\"].mean(skipna=True)\n    print(f\"[{split_name}] n={len(y)} | macro AUROC = {macro:.4f}\")\n    print(res.to_string(index=False, float_format=lambda v: f\"{v:.4f}\"))\n    return macro, res","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:46:17.70331Z","iopub.execute_input":"2026-10-02T17:46:17.704569Z","iopub.status.idle":"2026-10-02T17:46:17.732947Z","shell.execute_reply.started":"2026-10-02T17:46:17.704533Z","shell.execute_reply":"2026-10-02T17:46:17.732092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def run_train(cfg, index_csv, npy_path, label_cols):\n    torch.manual_seed(cfg.seed); np.random.seed(cfg.seed)\n    out_dir = Path(cfg.out_dir); out_dir.mkdir(parents=True, exist_ok=True)\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    print(\"[env]\", torch.cuda.get_device_name(0) if device.type == \"cuda\" else \"CPU\")\n\n    loaders = make_loaders(index_csv, npy_path, label_cols, cfg.img_size,\n                           cfg.batch_size, cfg.workers,\n                           balance_col=\"nodule_mass\" if label_cols == [\"nodule_mass\"] else None)\n\n    tr = pd.read_csv(index_csv).query(\"split == 'train'\")\n    pos = tr[label_cols].sum().to_numpy(float)\n    pos_weight = np.clip((len(tr) - pos) / np.maximum(pos, 1), 1.0, cfg.max_pos_weight)\n    criterion = nn.BCEWithLogitsLoss(\n        pos_weight=torch.tensor(pos_weight, dtype=torch.float32, device=device))\n\n    model = build_model(len(label_cols), cfg.model, cfg.weights_file,\n                        cfg.grad_ckpt, cfg.drop_path).to(device, memory_format=torch.channels_last)\n    opt = build_optimizer(model, cfg.lr, cfg.wd, cfg.layer_decay)\n    steps = len(loaders[\"train\"]) * cfg.epochs\n    sched = cosine_with_warmup(opt, int(cfg.warmup_frac * steps), steps)\n    scaler = torch.amp.GradScaler(\"cuda\", enabled=device.type == \"cuda\")\n\n    t0, best, history = time.time(), -1.0, []\n    for epoch in range(1, cfg.epochs + 1):\n        model.train(); running = seen = 0\n        for step, (x, y) in enumerate(loaders[\"train\"], 1):\n            x = x.to(device, non_blocking=True, memory_format=torch.channels_last)\n            y = y.to(device, non_blocking=True)\n            y = y[:, None] if y.ndim == 1 else y\n            with torch.autocast(\"cuda\", dtype=torch.float16, enabled=device.type == \"cuda\"):\n                loss = criterion(model(x), y)\n            opt.zero_grad(set_to_none=True)\n            scaler.scale(loss).backward(); scaler.unscale_(opt)\n            torch.nn.utils.clip_grad_norm_(model.parameters(), 5.0)\n            scaler.step(opt); scaler.update(); sched.step()\n            running += loss.item() * len(x); seen += len(x)\n            if step % 50 == 0:\n                print(f\"  ep{epoch} [{step}/{len(loaders['train'])}] loss {running/seen:.4f} \"\n                      f\"({(time.time()-t0)/60:.1f} phút)\")\n\n        macro, per_class = evaluate(model, loaders.get(\"val\", loaders[\"train\"]),\n                                    device, label_cols, f\"val@ep{epoch}\")\n        history.append(dict(epoch=epoch, train_loss=running/seen, val_macro_auroc=macro))\n        if macro > best:\n            best = macro\n            torch.save({\"model\": model.state_dict(), \"epoch\": epoch,\n                        \"label_cols\": label_cols}, out_dir / \"best.pt\")\n            per_class.to_csv(out_dir / \"val_per_class_best.csv\", index=False)\n            print(f\"  -> lưu checkpoint tốt nhất (macro AUROC {macro:.4f})\")\n        if (time.time() - t0) / 3600 > cfg.max_hours:\n            print(\"[stop] hết thời gian cho phép\"); break\n\n    hist = pd.DataFrame(history); hist.to_csv(out_dir / \"history.csv\", index=False)\n\n    test_res = None\n    if \"test\" in loaders:\n        ck = torch.load(out_dir / \"best.pt\", map_location=device, weights_only=False)\n        model.load_state_dict(ck[\"model\"])\n        _, test_res = evaluate(model, loaders[\"test\"], device, label_cols, \"test\")\n        test_res.to_csv(out_dir / \"test_per_class.csv\", index=False)\n    return hist, test_res","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:46:17.73455Z","iopub.execute_input":"2026-10-02T17:46:17.735266Z","iopub.status.idle":"2026-10-02T17:46:17.765368Z","shell.execute_reply.started":"2026-10-02T17:46:17.735219Z","shell.execute_reply":"2026-10-02T17:46:17.764504Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Kết quả sẽ được tách làm 2 nhóm:\n1. **Target (Nodule/Mass):** Bệnh lý trọng tâm.\n2. **Auxiliary:** 12 nhãn bệnh nền còn lại.","metadata":{}},{"cell_type":"code","source":"# Kích hoạt train ConvNeXt V2 trên tập dữ liệu đã gộp\nprint(\"🚀 Đang chạy Baseline ConvNeXt V2 trên tập NIH + VinBigData...\")\nhist, test_res = run_train(CFG, idx_path_mixed, npy_path_mixed, NIH_14_LABELS)\n\n# Tách điểm số đánh giá (Nodule/Mass vs Các bệnh khác)\nif test_res is not None:\n    print(\"\\n📊 PHÂN TÍCH CHUYÊN SÂU KẾT QUẢ BASELINE:\")\n    nodule_mass_idx = [test_res[test_res['label'] == c].index[0] for c in ['Nodule', 'Mass'] if c in test_res['label'].values]\n    aux_idx = [i for i in range(len(test_res)) if i not in nodule_mass_idx]\n    \n    nm_res = test_res.iloc[nodule_mass_idx]\n    aux_res = test_res.iloc[aux_idx]\n    \n    print(f\"-> Trung bình AP nhóm Target (Nodule/Mass): {nm_res['ap'].mean():.4f}\")\n    print(f\"-> Trung bình AP nhóm Auxiliary (12 bệnh phụ): {aux_res['ap'].mean():.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T17:46:17.766907Z","iopub.execute_input":"2026-10-02T17:46:17.767877Z","iopub.status.idle":"2026-10-02T18:23:36.68357Z","shell.execute_reply.started":"2026-10-02T17:46:17.767796Z","shell.execute_reply":"2026-10-02T18:23:36.678716Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Biểu đồ kết quả","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(14, 4.5))\n\nax2 = ax[0].twinx()\nax[0].plot(hist[\"epoch\"], hist[\"train_loss\"], \"o-\", color=\"#1f77b4\", label=\"train loss\")\nax2.plot(hist[\"epoch\"], hist[\"val_macro_auroc\"], \"s-\", color=\"#d62728\", label=\"val macro AUROC\")\nax[0].set_xlabel(\"epoch\"); ax[0].set_ylabel(\"loss\", color=\"#1f77b4\")\nax2.set_ylabel(\"macro AUROC\", color=\"#d62728\"); ax[0].set_title(\"Đường huấn luyện\")\n\nif test_res is not None:\n    r = test_res.dropna(subset=[\"auroc\"]).sort_values(\"auroc\")\n    colors = [\"#d62728\" if l in NSCLC_PRIMARY else \"#7f7f7f\" for l in r[\"label\"]]\n    ax[1].barh(r[\"label\"], r[\"auroc\"], color=colors)\n    ax[1].axvline(0.5, color=\"k\", ls=\"--\", lw=0.8)\n    ax[1].set_xlim(0.4, 1.0); ax[1].set_title(\"AUROC trên test (đỏ = Nodule/Mass)\")\n    ax[1].tick_params(axis=\"y\", labelsize=8)\n    for i, v in enumerate(r[\"auroc\"]):\n        ax[1].text(v, i, f\" {v:.3f}\", va=\"center\", fontsize=7)\n\nplt.tight_layout(); plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T18:23:36.689042Z","iopub.status.idle":"2026-10-02T18:23:36.689572Z","shell.execute_reply.started":"2026-10-02T18:23:36.689328Z","shell.execute_reply":"2026-10-02T18:23:36.689353Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phần 10 — MedNeXt-2D đa nhiệm","metadata":{}},{"cell_type":"code","source":"class GRN2d(nn.Module):\n    \"\"\"Global Response Normalization (ConvNeXt V2 / MedNeXt-v2), bản 2D.\"\"\"\n    def __init__(self, dim):\n        super().__init__()\n        self.gamma = nn.Parameter(torch.zeros(1, dim, 1, 1))\n        self.beta = nn.Parameter(torch.zeros(1, dim, 1, 1))\n\n    def forward(self, x):\n        gx = torch.norm(x, p=2, dim=(2, 3), keepdim=True)\n        nx = gx / (gx.mean(dim=1, keepdim=True) + 1e-6)\n        return self.gamma * (x * nx) + self.beta + x\n\n\nclass MedNeXtBlock2D(nn.Module):\n    \"\"\"Khối MedNeXt: depthwise conv -> InstanceNorm -> mở rộng 1x1 -> GELU -> GRN -> nén 1x1\n    -> residual (Roy et al., MICCAI 2023; GRN theo MedNeXt-v2). Bản trước đó thiếu\n    InstanceNorm và GRN nên thực chất là một khối ConvNeXt V1 thường, không phải MedNeXt.\"\"\"\n    def __init__(self, dim, exp_r=4, kernel_size=7):\n        super().__init__()\n        self.dwconv = nn.Conv2d(dim, dim, kernel_size, padding=kernel_size // 2, groups=dim)\n        self.norm = nn.InstanceNorm2d(dim, affine=True)\n        self.pwconv1 = nn.Conv2d(dim, dim * exp_r, 1)\n        self.act = nn.GELU()\n        self.grn = GRN2d(dim * exp_r)\n        self.pwconv2 = nn.Conv2d(dim * exp_r, dim, 1)\n\n    def forward(self, x):\n        h = self.pwconv1(self.norm(self.dwconv(x)))\n        return x + self.pwconv2(self.grn(self.act(h)))\n\n\nclass MedNeXtDownBlock2D(nn.Module):\n    \"\"\"Giảm 2x kích thước, đổi số kênh; nhánh residual cũng được resample bằng conv 1x1\n    stride 2 để khớp shape (residual-resampling, Roy et al. 2023).\"\"\"\n    def __init__(self, dim_in, dim_out, exp_r=4, kernel_size=7):\n        super().__init__()\n        self.dwconv = nn.Conv2d(dim_in, dim_in, kernel_size, stride=2,\n                                padding=kernel_size // 2, groups=dim_in)\n        self.norm = nn.InstanceNorm2d(dim_in, affine=True)\n        self.pwconv1 = nn.Conv2d(dim_in, dim_in * exp_r, 1)\n        self.act = nn.GELU()\n        self.grn = GRN2d(dim_in * exp_r)\n        self.pwconv2 = nn.Conv2d(dim_in * exp_r, dim_out, 1)\n        self.res = nn.Conv2d(dim_in, dim_out, 1, stride=2)\n\n    def forward(self, x):\n        h = self.pwconv1(self.norm(self.dwconv(x)))\n        return self.res(x) + self.pwconv2(self.grn(self.act(h)))\n\n\nclass MedNeXt2D_MultiTask(nn.Module):\n    \"\"\"Encoder 2D theo khối MedNeXt — TỰ CÀI LẠI cho phân loại/khu trú 2D, không phải kho mã\n    3D chính thức. Stem stride 4 (giống ConvNeXt) thay vì stride 1 của bản gốc (thiết kế cho\n    phân đoạn 3D, nơi giữ nguyên độ phân giải ở tầng đầu quan trọng hơn) để chi phí tính toán\n    khả thi ở ảnh 384px.\"\"\"\n    def __init__(self, in_chans=3, num_classes=14, depths=(3, 3, 9, 3),\n                dims=(96, 192, 384, 768), exp_r=4, kernel_size=7):\n        super().__init__()\n        self.stem = nn.Sequential(nn.Conv2d(in_chans, dims[0], kernel_size=4, stride=4),\n                                  nn.InstanceNorm2d(dims[0], affine=True))\n        self.stages = nn.ModuleList()\n        in_dim = dims[0]\n        for i, (depth, dim) in enumerate(zip(depths, dims)):\n            blocks = [MedNeXtDownBlock2D(in_dim, dim, exp_r, kernel_size)] if i > 0 else []\n            blocks += [MedNeXtBlock2D(dim, exp_r, kernel_size) for _ in range(depth)]\n            self.stages.append(nn.Sequential(*blocks))\n            in_dim = dim\n        self.norm_out = nn.InstanceNorm2d(dims[-1], affine=True)\n        self.class_head = nn.Linear(dims[-1], num_classes)\n        self.bbox_head = nn.Sequential(nn.Linear(dims[-1], 256), nn.GELU(),\n                                       nn.Linear(256, 4), nn.Sigmoid())\n\n    def forward_features(self, x):\n        x = self.stem(x)\n        for stage in self.stages:\n            x = stage(x)\n        return self.norm_out(x).mean(dim=(2, 3))\n\n    def forward(self, x):\n        feat = self.forward_features(x)\n        return self.class_head(feat), self.bbox_head(feat)\n\n\nclass ClsOnlyWrapper(nn.Module):\n    \"\"\"Bọc model đa nhiệm để TÁI DÙNG evaluate() đã viết cho ConvNeXt V2 ở Phần 9 — tránh\n    viết lại một bản evaluate() thứ hai gần như giống hệt chỉ để đổi chỗ lấy logits.\"\"\"\n    def __init__(self, model):\n        super().__init__()\n        self.model = model\n\n    def forward(self, x):\n        return self.model(x)[0]\n\n\nclass MultiTaskLoss(nn.Module):\n    def __init__(self, pos_weight=None, bbox_weight=1.0):\n        super().__init__()\n        self.cls_criterion = nn.BCEWithLogitsLoss(pos_weight=pos_weight)\n        self.bbox_criterion = nn.L1Loss(reduction=\"none\")\n        self.bbox_weight = bbox_weight\n\n    def forward(self, cls_logits, bbox_preds, cls_targets, bbox_targets, has_bbox):\n        cls_loss = self.cls_criterion(cls_logits, cls_targets)\n        l1 = self.bbox_criterion(bbox_preds, bbox_targets).mean(dim=1) * has_bbox\n        n = has_bbox.sum()\n        # has_bbox toàn 0 trong batch (rất có thể xảy ra vì chỉ ~vài % ảnh VinBig có box) ->\n        # vẫn phải trả một loss có đồ thị tính đạo hàm (0 * bbox_preds) để backward() không lỗi\n        bbox_loss = (l1.sum() / n) if n > 0 else (bbox_preds.sum() * 0.0)\n        return cls_loss + self.bbox_weight * bbox_loss, cls_loss, bbox_loss\n\n\nclass HarmonizedXRayDataset(CXRDataset):\n    \"\"\"Giống CXRDataset nhưng trả thêm (bbox, has_bbox). has_bbox suy từ toạ độ khác 0 —\n    ảnh NIH luôn là [0,0,0,0] (không có bbox), ảnh VinBig có box thật sau khi Fix ở Phần 4.\"\"\"\n    def __init__(self, index_df, npy_path, label_cols, img_size=384, train=False):\n        super().__init__(index_df, npy_path, label_cols, img_size, train)\n        self.bboxes = self.df[[\"x_min\", \"y_min\", \"x_max\", \"y_max\"]].to_numpy(np.float32)\n\n    def __getitem__(self, i):\n        img, label = super().__getitem__(i)\n        box = self.bboxes[i]\n        has_bbox = float((box[2] - box[0]) > 1e-4 and (box[3] - box[1]) > 1e-4)\n        return img, label, torch.tensor(box, dtype=torch.float32), torch.tensor(has_bbox)\n\n\ndef make_mednext_train_loader(index_csv, npy_path, label_cols, img_size, batch_size, workers):\n    \"\"\"DataLoader CÓ bbox, chỉ dùng cho tập train (multitask loss chỉ tính lúc train).\n    Val/test dùng make_loaders() như Phần 9 vì evaluate() chỉ cần (ảnh, nhãn).\"\"\"\n    df = pd.read_csv(index_csv)\n    df = df[df[\"cache_ok\"]] if \"cache_ok\" in df.columns else df\n    part = df[df[\"split\"] == \"train\"]\n    ds = HarmonizedXRayDataset(part, npy_path, label_cols, img_size, train=True)\n    loader = DataLoader(ds, batch_size=batch_size, shuffle=True, num_workers=workers,\n                        pin_memory=True, drop_last=True)\n    print(f\"[loader] train (đa nhiệm, có bbox): {len(ds)} ảnh\")\n    return loader\n\n\ndef run_train_mednext(cfg, index_csv, npy_path, label_cols, bbox_weight=1.0,\n                      out_dir=\"/kaggle/working/run_mednext\"):\n    \"\"\"Phỏng theo đúng cấu trúc run_train() ở Phần 9 (history.csv, best.pt,\n    test_per_class.csv cùng quy ước) để Phần 11 đọc được và so sánh công bằng.\"\"\"\n    torch.manual_seed(cfg.seed); np.random.seed(cfg.seed)\n    out_dir = Path(out_dir); out_dir.mkdir(parents=True, exist_ok=True)\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    print(\"[env]\", torch.cuda.get_device_name(0) if device.type == \"cuda\" else \"CPU\")\n\n    train_loader = make_mednext_train_loader(index_csv, npy_path, label_cols,\n                                             cfg.img_size, cfg.batch_size, cfg.workers)\n    eval_loaders = make_loaders(index_csv, npy_path, label_cols, cfg.img_size,\n                                cfg.batch_size, cfg.workers)\n\n    tr = pd.read_csv(index_csv).query(\"split == 'train'\")\n    pos = tr[label_cols].sum().to_numpy(float)\n    pos_weight = np.clip((len(tr) - pos) / np.maximum(pos, 1), 1.0, cfg.max_pos_weight)\n    criterion = MultiTaskLoss(\n        pos_weight=torch.tensor(pos_weight, dtype=torch.float32, device=device),\n        bbox_weight=bbox_weight)\n\n    model = MedNeXt2D_MultiTask(num_classes=len(label_cols)).to(device, memory_format=torch.channels_last)\n    print(f\"[model] MedNeXt-2D | {sum(p.numel() for p in model.parameters())/1e6:.1f}M tham số\")\n    opt = torch.optim.AdamW(model.parameters(), lr=cfg.lr, weight_decay=cfg.wd)\n    steps = len(train_loader) * cfg.epochs\n    sched = cosine_with_warmup(opt, int(cfg.warmup_frac * steps), steps)\n    scaler = torch.amp.GradScaler(\"cuda\", enabled=device.type == \"cuda\")\n    eval_model = ClsOnlyWrapper(model)\n\n    t0, best, history = time.time(), -1.0, []\n    for epoch in range(1, cfg.epochs + 1):\n        model.train()\n        run_total = run_cls = run_bbox = seen = 0\n        for step, (x, y, bbox, has_bbox) in enumerate(train_loader, 1):\n            x = x.to(device, non_blocking=True, memory_format=torch.channels_last)\n            y = y.to(device, non_blocking=True)\n            bbox = bbox.to(device, non_blocking=True)\n            has_bbox = has_bbox.to(device, non_blocking=True)\n            with torch.autocast(\"cuda\", dtype=torch.float16, enabled=device.type == \"cuda\"):\n                cls_logits, bbox_preds = model(x)\n                total, cls_loss, bbox_loss = criterion(cls_logits, bbox_preds, y, bbox, has_bbox)\n            opt.zero_grad(set_to_none=True)\n            scaler.scale(total).backward(); scaler.unscale_(opt)\n            torch.nn.utils.clip_grad_norm_(model.parameters(), 5.0)\n            scaler.step(opt); scaler.update(); sched.step()\n            run_total += total.item() * len(x); run_cls += cls_loss.item() * len(x)\n            run_bbox += bbox_loss.item() * len(x); seen += len(x)\n            if step % 50 == 0:\n                print(f\"  ep{epoch} [{step}/{len(train_loader)}] loss {run_total/seen:.4f} \"\n                      f\"(cls {run_cls/seen:.4f}, bbox {run_bbox/seen:.4f}) \"\n                      f\"({(time.time()-t0)/60:.1f} phút)\")\n\n        macro, per_class = evaluate(eval_model, eval_loaders.get(\"val\", eval_loaders[\"train\"]),\n                                    device, label_cols, f\"val@ep{epoch}\")\n        history.append(dict(epoch=epoch, train_loss=run_total / seen,\n                            train_bbox_loss=run_bbox / seen, val_macro_auroc=macro))\n        if macro > best:\n            best = macro\n            torch.save({\"model\": model.state_dict(), \"epoch\": epoch,\n                        \"label_cols\": label_cols}, out_dir / \"best.pt\")\n            per_class.to_csv(out_dir / \"val_per_class_best.csv\", index=False)\n            print(f\"  -> lưu checkpoint tốt nhất (macro AUROC {macro:.4f})\")\n        if (time.time() - t0) / 3600 > cfg.max_hours:\n            print(\"[stop] hết thời gian cho phép\"); break\n\n    hist = pd.DataFrame(history); hist.to_csv(out_dir / \"history.csv\", index=False)\n\n    test_res = None\n    if \"test\" in eval_loaders:\n        ck = torch.load(out_dir / \"best.pt\", map_location=device, weights_only=False)\n        model.load_state_dict(ck[\"model\"])\n        _, test_res = evaluate(eval_model, eval_loaders[\"test\"], device, label_cols, \"test\")\n        test_res.to_csv(out_dir / \"test_per_class.csv\", index=False)\n    return hist, test_res","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T18:23:36.693126Z","iopub.status.idle":"2026-10-02T18:23:36.693487Z","shell.execute_reply.started":"2026-10-02T18:23:36.693321Z","shell.execute_reply":"2026-10-02T18:23:36.693341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Đang chạy MedNeXt-2D đa nhiệm (phân loại + bbox) trên tập NIH + VinBigData...\")\nhist_med, test_res_med = run_train_mednext(CFG, idx_path_mixed, npy_path_mixed, NIH_14_LABELS)\n\nif test_res_med is not None:\n    print(\"\\nPHÂN TÍCH CHUYÊN SÂU KẾT QUẢ MEDNEXT-2D (cùng cách tách nhóm như Phần 9):\")\n    nm_idx = [test_res_med[test_res_med[\"label\"] == c].index[0]\n             for c in [\"Nodule\", \"Mass\"] if c in test_res_med[\"label\"].values]\n    aux_idx = [i for i in range(len(test_res_med)) if i not in nm_idx]\n    print(f\"-> Trung bình AP nhóm Target (Nodule/Mass): {test_res_med.iloc[nm_idx]['ap'].mean():.4f}\")\n    print(f\"-> Trung bình AP nhóm Auxiliary (12 bệnh phụ): {test_res_med.iloc[aux_idx]['ap'].mean():.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T18:23:36.695856Z","iopub.status.idle":"2026-10-02T18:23:36.696423Z","shell.execute_reply.started":"2026-10-02T18:23:36.696135Z","shell.execute_reply":"2026-10-02T18:23:36.696166Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Phần 11: So sánh 2 model trên ảnh X-quang","metadata":{}},{"cell_type":"code","source":"# Đường dẫn tới 2 file log\nconvnext_log_path = \"/kaggle/working/run_convnextv2/history.csv\"\nmednext_log_path = \"/kaggle/working/run_mednext/history.csv\"\n\nif os.path.exists(convnext_log_path) and os.path.exists(mednext_log_path):\n    # Đọc dữ liệu\n    df_conv = pd.read_csv(convnext_log_path)\n    df_med = pd.read_csv(mednext_log_path)\n\n    fig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\n    # 1. So sánh Train Classification Loss (Càng thấp càng tốt)\n    axes[0].plot(df_conv['epoch'], df_conv['train_loss'], label='ConvNeXt V2 (Baseline)', marker='o', color='#1f77b4')\n    axes[0].plot(df_med['epoch'], df_med['train_loss'], label='MedNeXt-2D (Đa nhiệm)', marker='s', color='#d62728')\n    axes[0].set_title(\"Train Classification Loss (Công bằng)\", fontsize=12)\n    axes[0].set_xlabel(\"Epoch\")\n    axes[0].set_ylabel(\"BCE Loss\")\n    axes[0].legend()\n    axes[0].grid(True, linestyle='--', alpha=0.6)\n\n    # 2. So sánh Validation Macro AUROC (Càng cao càng tốt)\n    axes[1].plot(df_conv['epoch'], df_conv['val_macro_auroc'], label='ConvNeXt V2 (Baseline)', marker='o', color='#1f77b4')\n    axes[1].plot(df_med['epoch'], df_med['val_macro_auroc'], label='MedNeXt-2D (Đa nhiệm)', marker='s', color='#d62728')\n    axes[1].set_title(\"Validation Macro AUROC\", fontsize=12)\n    axes[1].set_xlabel(\"Epoch\")\n    axes[1].set_ylabel(\"AUROC Score\")\n    axes[1].legend()\n    axes[1].grid(True, linestyle='--', alpha=0.6)\n\n    plt.suptitle(\"So sánh hiệu suất: Baseline vs Mô hình Đề xuất\", fontsize=14, fontweight='bold')\n    plt.tight_layout()\n    plt.show()\nelse:\n    print(\"Chưa tìm thấy đủ 2 file history.csv. Hãy đảm bảo bạn đã train xong cả 2 mô hình!\")\n\n# So sánh TÁCH RIÊNG Nodule/Mass (target) và 12 nhãn phụ (auxiliary) trên tập test —\n# đọc lại 2 file test_per_class.csv mà run_train()/run_train_mednext() đã lưu sẵn.\nconvnext_test_path = \"/kaggle/working/run_convnextv2/test_per_class.csv\"\nmednext_test_path = \"/kaggle/working/run_mednext/test_per_class.csv\"\n\nif os.path.exists(convnext_test_path) and os.path.exists(mednext_test_path):\n    def group_ap(path):\n        r = pd.read_csv(path)\n        nm = r[r[\"label\"].isin([\"Nodule\", \"Mass\"])][\"ap\"].mean()\n        aux = r[~r[\"label\"].isin([\"Nodule\", \"Mass\"])][\"ap\"].mean()\n        return nm, aux\n\n    nm_c, aux_c = group_ap(convnext_test_path)\n    nm_m, aux_m = group_ap(mednext_test_path)\n\n    fig, ax = plt.subplots(figsize=(6, 4.5))\n    x, w = np.arange(2), 0.35\n    ax.bar(x - w / 2, [nm_c, aux_c], w, label=\"ConvNeXt V2\", color=\"#1f77b4\")\n    ax.bar(x + w / 2, [nm_m, aux_m], w, label=\"MedNeXt-2D\", color=\"#d62728\")\n    ax.set_xticks(x); ax.set_xticklabels([\"Nodule/Mass\\n(target)\", \"12 nhãn phụ\\n(auxiliary)\"])\n    ax.set_ylabel(\"Average Precision (test)\")\n    ax.set_title(\"So sánh tách nhóm: target vs auxiliary\")\n    ax.legend(); plt.tight_layout(); plt.show()\n\n    print(f\"ConvNeXt V2  — target AP {nm_c:.4f} | auxiliary AP {aux_c:.4f}\")\n    print(f\"MedNeXt-2D   — target AP {nm_m:.4f} | auxiliary AP {aux_m:.4f}\")\nelse:\n    print(\"Chưa tìm thấy đủ 2 file test_per_class.csv — chạy xong cả 2 model ở Phần 9 và 10 trước.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-02T18:23:36.698496Z","iopub.status.idle":"2026-10-02T18:23:36.698904Z","shell.execute_reply.started":"2026-10-02T18:23:36.698751Z","shell.execute_reply":"2026-10-02T18:23:36.698773Z"}},"outputs":[],"execution_count":null}]}