{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":15094918,"datasetId":9664561,"databundleVersionId":15979687}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pydicom\nfrom pydicom.multival import MultiValue\nfrom pydicom.pixel_data_handlers.util import apply_modality_lut, apply_voi_lut\nimport numpy as np\nimport cv2\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\n\n# =========================\n# PATH CONFIG\n# =========================\nDICOM_DIR = \"/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection/train\"\nANNOT_PATH = \"/kaggle/input/datasets/benxelua/correct-label/annotations/annotations_train.csv\"\nIMAGE_LABEL_PATH = \"/kaggle/input/datasets/benxelua/correct-label/annotations/image_labels_train.csv\"\n\n# =========================\n# LOAD CSV\n# =========================\ndf_annot = pd.read_csv(ANNOT_PATH)\ndf_labels = pd.read_csv(IMAGE_LABEL_PATH)\n\n# =========================\n# PREPROCESS CONFIG\n# =========================\nLOW_PERCENTILE = 0.5\nHIGH_PERCENTILE = 99.5\n\n# =========================\n# BASIC FUNCTIONS\n# =========================\n\ndef normalize_8bit(image):\n    img = np.asarray(image, dtype=np.float32)\n    img = np.nan_to_num(img, nan=0.0, posinf=0.0, neginf=0.0)\n    img_min, img_max = float(np.min(img)), float(np.max(img))\n    if img_max > img_min:\n        img = (img - img_min) / (img_max - img_min)\n    else:\n        img = np.zeros_like(img, dtype=np.float32)\n    return (img * 255.0).clip(0, 255).astype(np.uint8)\n\n\ndef fix_monochrome(img, ds):\n    if getattr(ds, \"PhotometricInterpretation\", \"\") == \"MONOCHROME1\":\n        img = np.max(img) - img\n    return img\n\n\ndef dicom_rescale_pixels(ds):\n    img = ds.pixel_array.astype(np.float32)\n    slope = float(getattr(ds, \"RescaleSlope\", 1.0))\n    intercept = float(getattr(ds, \"RescaleIntercept\", 0.0))\n    return img * slope + intercept\n\n\ndef first_float(value, default):\n    if value is None:\n        return float(default)\n    if hasattr(value, \"value\"):\n        value = value.value\n    if isinstance(value, (MultiValue, list, tuple, np.ndarray)):\n        if len(value) == 0:\n            return float(default)\n        value = value[0]\n    try:\n        return float(value)\n    except Exception:\n        return float(default)\n\n\n# ──────────────────────────────────────────────────────────────────────\n# Các hàm xử lý từng kênh — theo lung-yolo-mutiple-preprocess.ipynb\n# ──────────────────────────────────────────────────────────────────────\n\ndef raw_minmax_from_ds(ds):\n    img = ds.pixel_array.astype(np.float32)\n    img = fix_monochrome(img, ds)\n    return normalize_8bit(img)\n\n\ndef voi_lut_from_ds(ds):\n    img = apply_modality_lut(ds.pixel_array, ds)\n    img = apply_voi_lut(img, ds).astype(np.float32)\n    img = fix_monochrome(img, ds)\n    return normalize_8bit(img)\n\n\ndef percentile_from_ds(ds, low=LOW_PERCENTILE, high=HIGH_PERCENTILE):\n    img = dicom_rescale_pixels(ds)\n    img = fix_monochrome(img, ds)\n    lo = float(np.percentile(img, low))\n    hi = float(np.percentile(img, high))\n    if hi <= lo:\n        hi = lo + 1.0\n    img = np.clip(img, lo, hi)\n    img = (img - lo) / (hi - lo)\n    return (img * 255.0).clip(0, 255).astype(np.uint8)\n\n\ndef clahe_from_ds(ds):\n    img = dicom_rescale_pixels(ds)\n    img = fix_monochrome(img, ds)\n\n    default_center = float(np.median(img))\n    default_width = float(np.std(img) * 4.0)\n    c_orig = first_float(ds.get(\"WindowCenter\", default_center), default_center)\n    w_orig = first_float(ds.get(\"WindowWidth\", default_width), default_width)\n\n    if w_orig < 1:\n        w_orig = float(np.std(img) * 4.0 + 1e-6)\n\n    img_min = c_orig - w_orig / 2.0\n    img_max = c_orig + w_orig / 2.0\n    if img_max <= img_min:\n        img_min = float(np.min(img))\n        img_max = float(np.max(img))\n        if img_max <= img_min:\n            return np.zeros_like(img, dtype=np.uint8)\n\n    img_windowed = np.clip(img, img_min, img_max)\n    img_8bit = ((img_windowed - img_min) / (img_max - img_min) * 255.0).clip(0, 255).astype(np.uint8)\n\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    return clahe.apply(img_8bit)\n\n\n# =========================\n# LABEL + BBOX\n# =========================\n\ndef get_image_labels(image_id):\n    row = df_labels[df_labels['image_id'] == image_id]\n    if row.empty:\n        return \"No Finding\"\n    labels = row.columns[(row == 1).iloc[0]].tolist()\n    labels = [l for l in labels if l != \"image_id\"]\n    return \", \".join(labels) if labels else \"No Finding\"\n\n\ndef draw_annotations(image, bboxes, class_names):\n    img = normalize_8bit(image)\n    img = cv2.cvtColor(img, cv2.COLOR_GRAY2RGB)\n\n    unique_classes = sorted(set(class_names))\n    color_map = {}\n\n    np.random.seed(42)  # để màu cố định mỗi lần chạy\n    for cls in unique_classes:\n        color_map[cls] = tuple(np.random.randint(0, 255, 3).tolist())\n\n    # vẽ bbox\n    for box, name in zip(bboxes, class_names):\n        if np.isnan(box).any():\n            continue\n\n        x1, y1, x2, y2 = map(int, box)\n        color = color_map[name]\n\n        cv2.rectangle(img, (x1, y1), (x2, y2), color, 3)\n\n    return img\n\n\n# =========================\n# MULTI-CHANNEL PROCESSING\n# =========================\n\ndef process_multi_channel(image_id):\n    path = os.path.join(DICOM_DIR, f\"{image_id}.dicom\")\n    ds = pydicom.dcmread(path)\n\n    channels = {}\n    channels['raw_minmax']  = raw_minmax_from_ds(ds)\n    channels['voi_lut']     = voi_lut_from_ds(ds)\n    channels['percentile']  = percentile_from_ds(ds)\n    channels['clahe']       = clahe_from_ds(ds)\n    channels['hist_eq']     = cv2.equalizeHist(channels['raw_minmax'])\n\n    res = {}\n    res['raw_minmax']  = channels['raw_minmax']\n    res['voi_lut']     = channels['voi_lut']\n    res['percentile']  = channels['percentile']\n    res['clahe']       = channels['clahe']\n\n    res['raw + voi + percentile'] = np.stack([\n        channels['raw_minmax'], channels['voi_lut'], channels['percentile']\n    ], axis=-1)\n\n    res['raw + voi + clahe'] = np.stack([\n        channels['raw_minmax'], channels['voi_lut'], channels['clahe']\n    ], axis=-1)\n\n    res['raw + percentile + clahe'] = np.stack([\n        channels['raw_minmax'], channels['percentile'], channels['clahe']\n    ], axis=-1)\n\n    res['voi + percentile + clahe'] = np.stack([\n        channels['voi_lut'], channels['percentile'], channels['clahe']\n    ], axis=-1)\n\n    res['hist_eq'] = channels['hist_eq']\n    res['adaptive'] = cv2.addWeighted(\n        channels['clahe'], 0.7,\n        channels['hist_eq'], 0.3, 0\n    )\n\n    return res\n\ndef save_preprocessing_images(results, image_id, output_dir=\"preprocess_outputs\"):\n    os.makedirs(output_dir, exist_ok=True)\n\n    for name, img in results.items():\n        # sanitize tên file\n        safe_name = name.replace(\" \", \"_\").replace(\"+\", \"plus\")\n\n        filename = f\"{image_id}_{safe_name}.png\"\n        path = os.path.join(output_dir, filename)\n\n        # xử lý multi-channel\n        if img.ndim == 3 and img.shape[-1] == 3:\n            save_img = img\n        else:\n            save_img = img\n\n        cv2.imwrite(path, save_img)\n\n    print(f\"Saved {len(results)} images to {output_dir}\")\n\n\n# ═════════════════════════════════════════════════════════════\n# PLOT 1: DICOM + BBOX + LABEL  (5 ảnh gốc)\n# ═════════════════════════════════════════════════════════════\n\ndef plot_dicom_grid(image_ids):\n    fig, axes = plt.subplots(1, 4, figsize=(22, 5), dpi=150)\n\n    for i, image_id in enumerate(image_ids):\n        path = os.path.join(DICOM_DIR, f\"{image_id}.dicom\")\n        ds = pydicom.dcmread(path)\n        raw = ds.pixel_array.astype(np.float32)\n        if getattr(ds, \"PhotometricInterpretation\", \"\") == \"MONOCHROME1\":\n            raw = np.max(raw) - raw\n        if hasattr(ds, 'RescaleSlope') and hasattr(ds, 'RescaleIntercept'):\n            raw = raw * ds.RescaleSlope + ds.RescaleIntercept\n\n        ann = df_annot[df_annot['image_id'] == image_id]\n        bboxes = ann[['x_min','y_min','x_max','y_max']].values\n        names = ann['class_name'].values\n\n        img = draw_annotations(raw, bboxes, names)\n        label = get_image_labels(image_id)\n\n        ax = axes[i]\n        ax.imshow(img)\n        ax.set_title(image_id[:6], fontsize=10)\n        # ax.text(0.5, -0.2, label, fontsize=9, ha='center',\n        #         transform=ax.transAxes, wrap=True)\n        ax.axis('off')\n\n    # plt.suptitle(\"Plot 1: DICOM samples with bounding boxes\", fontsize=14, fontweight='bold')\n    plt.tight_layout()\n    plt.savefig(\"plot1_dicom_bbox_label.pdf\", dpi=300, bbox_inches='tight')\n    plt.show()\n\n\n# ═════════════════════════════════════════════════════════════\n# PLOT 2: PREPROCESS GRID  (ảnh so sánh các phương pháp)\n# ═════════════════════════════════════════════════════════════\n\ndef plot_preprocessing(results):\n    methods = [\n        # 'raw_minmax',\n        # 'voi_lut',\n        # 'percentile',\n        # 'clahe',\n        # 'hist_eq',\n        # 'adaptive',\n        'raw + voi + percentile',\n        'raw + voi + clahe',\n        'raw + percentile + clahe',\n        'voi + percentile + clahe'\n    ]\n\n    fig, axes = plt.subplots(1, 4, figsize=(15, 6), dpi=150)\n    axes = axes.flatten()\n\n    for i, name in enumerate(methods):\n        img = results[name]\n        ax = axes[i]\n\n        if img.ndim == 3 and img.shape[-1] == 3:\n            ax.imshow(img)\n        else:\n            ax.imshow(img, cmap='gray')\n\n        ax.set_title(name, fontsize=15, fontweight='bold')\n        ax.axis('off')\n\n    plt.tight_layout(pad=1.0)\n    plt.subplots_adjust(wspace=0.1, hspace=0.1)\n    plt.savefig(\"plot2_preprocessing.pdf\", dpi=300, bbox_inches='tight')\n    plt.show()\n\n\n# ═════════════════════════════════════════════════════════════\n# PLOT 3: HISTOGRAM SO SÁNH\n# ═════════════════════════════════════════════════════════════\n\ndef plot_images(image_id):\n    \"\"\"\n    Tự động đọc, tiền xử lý và hiển thị các ảnh single-channel dựa vào image_id\n    \"\"\"\n    results = process_multi_channel(image_id)\n    \n    methods = [\n        'raw_minmax',\n        'voi_lut',\n        'percentile',\n        'clahe',\n        'hist_eq',\n        'adaptive'\n    ]\n    \n    # Map tên hiển thị đẹp cho matplotlib\n    display_names = {\n        'raw_minmax': 'Raw Min-max',\n        'voi_lut': 'VOI LUT',\n        'percentile': 'Percentile',\n        'clahe': 'CLAHE',\n        'hist_eq': 'Histogram Eq',\n        'adaptive': 'Adaptive Percentile'\n    }\n    \n    single = {k: results[k] for k in methods if k in results}\n\n    n = len(single)\n    if n == 0:\n        print(f\"Không có ảnh single-channel nào cho image_id: {image_id}\")\n        return\n\n    # Chỉ tạo 1 hàng (1 row), chiều cao 5\n    fig, axes = plt.subplots(1, n, figsize=(4.5 * n, 5), dpi=150)\n    \n    if n == 1:\n        axes = [axes]\n\n    for i, (name, img) in enumerate(single.items()):\n        ax_img = axes[i]\n        \n        # Plot ảnh\n        ax_img.imshow(img, cmap='gray', vmin=0, vmax=255)\n        \n        # Thêm lại dòng set_title bị thiếu\n        ax_img.set_title(display_names[name], fontsize=18, fontweight='bold')\n        ax_img.axis('off')\n\n    plt.tight_layout()\n    \n    # Lưu file tự động theo image_id\n    save_name = f\"plot3_single_preprocess.pdf\"\n    plt.savefig(save_name, dpi=300, bbox_inches='tight')\n    plt.show()\n\n\n# ═════════════════════════════════════════════════════════════\n# RUN\n# ═════════════════════════════════════════════════════════════\n\nimage_list = [\n    \"000d68e42b71d3eac10ccc077aba07c1\",\n    \"00150343289f317a0ad5629d5b7d9ef9\",\n    \"0061cf6d35e253b6e7f03940592cc35e\",\n    \"01570ee44031e4ebab6031501293bf66\"\n    # \"010018c93ed33ae56ed048ee54867e46\"\n]\n\nprocess_list = [\n    \"000d68e42b71d3eac10ccc077aba07c1\",\n    \"001d127bad87592efe45a5c7678f8b8d\",\n    \"0061cf6d35e253b6e7f03940592cc35e\",\n    \"010018c93ed33ae56ed048ee54867e46\",\n    \"0007d316f756b3fa0baea2ff514ce945\",\n    \"00150343289f317a0ad5629d5b7d9ef9\",\n    \"0046f681f078851293c4e710c4466058\",\n    \"00675cd546313f912cadd4ad54415d69\",\n    \"00bcb82818ea83d6a86df241762cd7d0\",\n    \"011ae9520e81f1efe71c9d954ec07d09\",\n    \"013893a5fa90241c65c3efcdbdd2cec1\",\n    \"01570ee44031e4ebab6031501293bf66\",\n    \"01546d3e6175ceaabd7d92f0c566579d\"\n]\n\n# Plot 1: DICOM + BBOX\nplot_dicom_grid(image_list)\n\n# # Plot 2: Preprocessing comparison\nresults = process_multi_channel(image_list[0])\nplot_preprocessing(results)\n\nfor image_id in process_list:\n    print(f\"Processing {image_id}...\")\n\n    results = process_multi_channel(image_id)\n    save_preprocessing_images(results, image_id)\n\nsave_preprocessing_images(results, image_list[0])\n\n# Plot 3: Histogram\nplot_images(\"000d68e42b71d3eac10ccc077aba07c1\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-12T02:35:55.156811Z","iopub.execute_input":"2026-05-12T02:35:55.157262Z","iopub.status.idle":"2026-05-12T02:37:41.509185Z","shell.execute_reply.started":"2026-05-12T02:35:55.157227Z","shell.execute_reply":"2026-05-12T02:37:41.507495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import shutil\n\n# def zip_preprocessing_outputs(folder=\"/kaggle/working/preprocess_outputs\", zip_name=\"preprocessing_images\"):\n#     shutil.make_archive(zip_name, 'zip', folder)\n#     print(f\"Created {zip_name}.zip\")\n\n# save_preprocessing_images(results, image_list[0])\n\n# # 🔥 ZIP\n# zip_preprocessing_outputs()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-12T02:37:41.511114Z","iopub.execute_input":"2026-05-12T02:37:41.51145Z","iopub.status.idle":"2026-05-12T02:37:41.516025Z","shell.execute_reply.started":"2026-05-12T02:37:41.511418Z","shell.execute_reply":"2026-05-12T02:37:41.515261Z"}},"outputs":[],"execution_count":null}]}