{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":14764895,"datasetId":9437596,"databundleVersionId":15616426}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport pydicom\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport concurrent.futures\n\n# =========================\n# CONFIG\n# =========================\nINPUT_DIR = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/test\"\nANNOTATION_CSV = \"/kaggle/input/datasets/tuktuai/vindr-physionet-384/annotations_test.csv\"\n\nOUTPUT_IMG_DIR = \"/kaggle/working/test_png\"\nOUTPUT_ANN_CSV = \"/kaggle/working/test_bbox.csv\"\n\nIMG_SIZE = 518   # \"same\" OR int (e.g., 640)\nLOW_PERCENTILE = 1\nHIGH_PERCENTILE = 99\nMAX_WORKERS = 4\n\nos.makedirs(OUTPUT_IMG_DIR, exist_ok=True)\n\n# =========================\n# IMAGE PREPROCESS\n# =========================\ndef dicom_to_uint8_rgb(dicom_path, img_size=\"same\", low_p=1, high_p=99):\n    dicom = pydicom.dcmread(dicom_path)\n    pixel_array = dicom.pixel_array.astype(np.float32)\n\n    # Handle MONOCHROME1\n    photometric = getattr(dicom, \"PhotometricInterpretation\", \"\")\n    if photometric == \"MONOCHROME1\":\n        pixel_array = np.max(pixel_array) - pixel_array\n\n    orig_h, orig_w = pixel_array.shape[:2]\n\n    # Percentile clipping\n    lo = np.percentile(pixel_array, low_p)\n    hi = np.percentile(pixel_array, high_p)\n\n    if hi <= lo:\n        lo = pixel_array.min()\n        hi = pixel_array.max()\n\n    pixel_array = np.clip(pixel_array, lo, hi)\n\n    # Normalize to [0,1]\n    if pixel_array.max() > pixel_array.min():\n        pixel_array = (pixel_array - pixel_array.min()) / (pixel_array.max() - pixel_array.min())\n    else:\n        pixel_array = np.zeros_like(pixel_array, dtype=np.float32)\n\n    pixel_array = (pixel_array * 255.0).clip(0, 255).astype(np.uint8)\n\n    # Resize only if needed\n    if img_size != \"same\":\n        img_processed = cv2.resize(pixel_array, (img_size, img_size), interpolation=cv2.INTER_AREA)\n    else:\n        img_processed = pixel_array\n\n    # Grayscale → RGB\n    img_rgb = np.stack([img_processed] * 3, axis=-1)\n\n    return img_rgb, orig_h, orig_w\n\n\ndef convert_one_image(filename):\n    if not filename.endswith(\".dicom\"):\n        return None\n\n    image_id = filename.replace(\".dicom\", \"\")\n    dicom_path = os.path.join(INPUT_DIR, filename)\n    save_path = os.path.join(OUTPUT_IMG_DIR, image_id + \".png\")\n\n    try:\n        # Resume logic\n        if os.path.exists(save_path):\n            dicom = pydicom.dcmread(dicom_path)\n            h, w = dicom.pixel_array.shape[:2]\n            return image_id, h, w\n\n        img_rgb, orig_h, orig_w = dicom_to_uint8_rgb(\n            dicom_path,\n            img_size=IMG_SIZE,\n            low_p=LOW_PERCENTILE,\n            high_p=HIGH_PERCENTILE\n        )\n\n        # Save (convert RGB → BGR for OpenCV)\n        img_bgr = cv2.cvtColor(img_rgb, cv2.COLOR_RGB2BGR)\n        cv2.imwrite(save_path, img_bgr)\n\n        return image_id, orig_h, orig_w\n\n    except Exception as e:\n        print(f\"Error converting {filename}: {e}\")\n        return None\n\n\n# =========================\n# STEP 1: CONVERT IMAGES\n# =========================\nfiles = os.listdir(INPUT_DIR)\nsize_map = {}\n\nwith concurrent.futures.ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:\n    results = list(tqdm(executor.map(convert_one_image, files), total=len(files)))\n\nfor item in results:\n    if item is not None:\n        image_id, h, w = item\n        size_map[image_id] = (h, w)\n\nprint(f\"Converted / indexed {len(size_map)} images.\")\n\n\n# =========================\n# STEP 2: UPDATE BBOX\n# =========================\ndf = pd.read_csv(ANNOTATION_CSV)\n\n# Clean column names\ndf.columns = [c.strip() for c in df.columns]\n\nrequired_cols = [\"image_id\", \"class_name\"]\nfor c in required_cols:\n    if c not in df.columns:\n        raise ValueError(f\"Missing required column: {c}\")\n\nbbox_cols = [\"x_min\", \"y_min\", \"x_max\", \"y_max\"]\nfor c in bbox_cols:\n    if c not in df.columns:\n        df[c] = np.nan\n\n\ndef scale_bbox_row(row):\n    # ✅ If keeping original size → no scaling\n    if IMG_SIZE == \"same\":\n        return row\n\n    image_id = row[\"image_id\"]\n\n    if image_id not in size_map:\n        return row\n\n    if row[\"class_name\"] == \"No finding\":\n        return row\n\n    if pd.isna(row[\"x_min\"]) or pd.isna(row[\"y_min\"]) or pd.isna(row[\"x_max\"]) or pd.isna(row[\"y_max\"]):\n        return row\n\n    orig_h, orig_w = size_map[image_id]\n\n    scale_x = IMG_SIZE / orig_w\n    scale_y = IMG_SIZE / orig_h\n\n    row[\"x_min\"] *= scale_x\n    row[\"x_max\"] *= scale_x\n    row[\"y_min\"] *= scale_y\n    row[\"y_max\"] *= scale_y\n\n    # Clamp\n    row[\"x_min\"] = np.clip(row[\"x_min\"], 0, IMG_SIZE - 1)\n    row[\"x_max\"] = np.clip(row[\"x_max\"], 0, IMG_SIZE - 1)\n    row[\"y_min\"] = np.clip(row[\"y_min\"], 0, IMG_SIZE - 1)\n    row[\"y_max\"] = np.clip(row[\"y_max\"], 0, IMG_SIZE - 1)\n\n    return row\n\n\ndf = df.apply(scale_bbox_row, axis=1)\ndf.to_csv(OUTPUT_ANN_CSV, index=False)\n\nprint(f\"Saved annotations to: {OUTPUT_ANN_CSV}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-28T01:53:50.619053Z","iopub.execute_input":"2025-11-28T01:53:50.619793Z","iopub.status.idle":"2025-11-28T01:54:08.397111Z","shell.execute_reply.started":"2025-11-28T01:53:50.619768Z","shell.execute_reply":"2025-11-28T01:54:08.395807Z"}},"outputs":[],"execution_count":null}]}