{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":24800,"datasetId":1042002,"databundleVersionId":1831594},{"sourceType":"datasetVersion","sourceId":1810938,"datasetId":1075803,"databundleVersionId":1848422}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport os\nimport cv2\nimport glob\n\norig_comp_dir = '/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection'\nresized_dataset_dir = '/kaggle/input/datasets/xhlulu/vinbigdata-chest-xray-resized-png-1024x1024'\n\ntrain_csv_path = os.path.join(orig_comp_dir, 'train.csv')\ntrain_meta_path = os.path.join(resized_dataset_dir, 'train_meta.csv')\n\ndf = pd.read_csv(train_csv_path)\n\nif os.path.exists(train_meta_path):\n    meta_df = pd.read_csv(train_meta_path)\n    df = df.merge(meta_df, on='image_id', how='left')\n    \n    TARGET_SIZE = 1024\n    df['x_min'] = df['x_min'] * (TARGET_SIZE / df['dim1'])\n    df['x_max'] = df['x_max'] * (TARGET_SIZE / df['dim1'])\n    df['y_min'] = df['y_min'] * (TARGET_SIZE / df['dim0'])\n    df['y_max'] = df['y_max'] * (TARGET_SIZE / df['dim0'])\n\nplt.figure(figsize=(14, 6))\nsns.countplot(data=df, y='class_name', order=df['class_name'].value_counts().index, palette='viridis')\nplt.title('Distribution of All Classes (including No finding)')\nplt.xlabel('Count')\nplt.ylabel('Class Name')\nplt.show()\n\ndf_abnormal = df[df['class_id'] != 14].copy()\n\nplt.figure(figsize=(14, 6))\nsns.countplot(data=df_abnormal, y='class_name', order=df_abnormal['class_name'].value_counts().index, palette='magma')\nplt.title('Distribution of Abnormalities (Excluding Class 14)')\nplt.xlabel('Count')\nplt.ylabel('Class Name')\nplt.show()\n\nfindings_per_image = df_abnormal.groupby('image_id').size()\nplt.figure(figsize=(10, 5))\nsns.histplot(findings_per_image, bins=30, kde=True, color='teal')\nplt.title('Number of Bounding Boxes per Image (Abnormal Images Only)')\nplt.xlabel('Number of Bounding Boxes')\nplt.ylabel('Frequency')\nplt.show()\n\ndf_abnormal['box_width'] = df_abnormal['x_max'] - df_abnormal['x_min']\ndf_abnormal['box_height'] = df_abnormal['y_max'] - df_abnormal['y_min']\ndf_abnormal['box_area'] = df_abnormal['box_width'] * df_abnormal['box_height']\n\nplt.figure(figsize=(12, 8))\nsns.boxplot(data=df_abnormal, x='box_area', y='class_name', order=df_abnormal.groupby('class_name')['box_area'].median().sort_values().index, palette='Set2')\nplt.title('Bounding Box Area Distribution per Class (Scaled to 1024x1024)')\nplt.xscale('log')\nplt.xlabel('Bounding Box Area (Log Scale)')\nplt.ylabel('Class Name')\nplt.show()\n\nplt.figure(figsize=(10, 10))\nsns.scatterplot(data=df_abnormal, x='box_width', y='box_height', hue='class_name', alpha=0.3, palette='tab20', s=10)\nplt.title('Bounding Box Width vs Height (Scaled to 1024x1024)')\nplt.xlabel('Width')\nplt.ylabel('Height')\nplt.legend(bbox_to_anchor=(1.05, 1), loc='upper left')\nplt.tight_layout()\nplt.show()\n\ndf_abnormal['x_center'] = (df_abnormal['x_min'] + df_abnormal['x_max']) / 2\ndf_abnormal['y_center'] = (df_abnormal['y_min'] + df_abnormal['y_max']) / 2\n\nplt.figure(figsize=(10, 10))\nsns.kdeplot(x=df_abnormal['x_center'], y=df_abnormal['y_center'], cmap='Reds', fill=True, bw_adjust=0.5)\nplt.title('Heatmap of Abnormality Locations (Center Coordinates)')\nplt.xlabel('X Coordinate')\nplt.ylabel('Y Coordinate')\nplt.gca().invert_yaxis()\nplt.show()\n\nrad_counts = df.groupby('rad_id').size().sort_values(ascending=False)\nplt.figure(figsize=(12, 5))\nsns.barplot(x=rad_counts.index, y=rad_counts.values, palette='Blues_r')\nplt.title('Number of Annotations per Radiologist')\nplt.xlabel('Radiologist ID')\nplt.ylabel('Number of Annotations')\nplt.xticks(rotation=45)\nplt.show()\n\nimage_dir = os.path.join(resized_dataset_dir, 'vinbigdata/train')\n\nif os.path.exists(image_dir):\n    sample_images = glob.glob(os.path.join(image_dir, '*.png'))\n    np.random.seed(42)\n    sample_images = np.random.choice(sample_images, min(200, len(sample_images)), replace=False)\n\n    laplacian_vars = []\n    clipped_ratios = []\n    histograms = np.zeros((256,))\n\n    for path in sample_images:\n        img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n        if img is None:\n            continue\n\n        laplacian_var = cv2.Laplacian(img, cv2.CV_64F).var()\n        laplacian_vars.append(laplacian_var)\n\n        clipped = np.sum(img == 0) + np.sum(img == 255)\n        total_pixels = img.shape[0] * img.shape[1]\n        clipped_ratios.append(clipped / total_pixels)\n\n        hist, _ = np.histogram(img.flatten(), 256, [0, 256])\n        histograms += hist\n\n    plt.figure(figsize=(18, 5))\n\n    plt.subplot(1, 3, 1)\n    sns.histplot(laplacian_vars, kde=True, color='purple')\n    plt.title('Image Sharpness (Laplacian Variance)')\n    plt.xlabel('Variance (Lower = Blurrier)')\n\n    plt.subplot(1, 3, 2)\n    sns.histplot(clipped_ratios, kde=True, color='red')\n    plt.title('Clipped Pixels Ratio (0 or 255)')\n    plt.xlabel('Ratio (Higher = Detail Loss)')\n\n    plt.subplot(1, 3, 3)\n    plt.plot(histograms, color='black')\n    plt.title('Aggregate Pixel Intensity Histogram')\n    plt.xlabel('Pixel Intensity')\n    plt.ylabel('Frequency')\n\n    plt.tight_layout()\n    plt.show()\n\n    fig, axes = plt.subplots(1, 4, figsize=(20, 5))\n    for ax, img_path in zip(axes, sample_images[:4]):\n        img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)\n        ax.imshow(img, cmap='gray')\n        ax.set_title(os.path.basename(img_path))\n        ax.axis('off')\n    plt.show()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T14:23:56.794853Z","iopub.execute_input":"2026-05-10T14:23:56.795895Z","iopub.status.idle":"2026-05-10T14:24:34.302192Z","shell.execute_reply.started":"2026-05-10T14:23:56.7958Z","shell.execute_reply":"2026-05-10T14:24:34.300942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport sys\nimport subprocess\nimport yaml\nimport random\nimport shutil\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom collections import defaultdict\nfrom PIL import Image\nfrom sklearn.model_selection import StratifiedGroupKFold\n\ntry:\n    from ensemble_boxes import weighted_boxes_fusion\nexcept ImportError:\n    subprocess.check_call([sys.executable, \"-m\", \"pip\", \"install\", \"ensemble-boxes\"])\n    from ensemble_boxes import weighted_boxes_fusion\n\nCLASS_SAMPLE_WEIGHTS = {\n    0: 1.0, 1: 2.8, 2: 2.2, 3: 1.1, 4: 2.5, 5: 2.4, 6: 2.0,\n    7: 1.6, 8: 1.5, 9: 1.5, 10: 1.5, 11: 1.1, 12: 3.5, 13: 1.1\n}\n\nCLASS_NAMES_14 = [\n    \"Aortic enlargement\", \"Atelectasis\", \"Calcification\",\n    \"Cardiomegaly\", \"Consolidation\", \"ILD\", \"Infiltration\",\n    \"Lung Opacity\", \"Nodule/Mass\", \"Other lesion\",\n    \"Pleural effusion\", \"Pleural thickening\", \"Pneumothorax\",\n    \"Pulmonary fibrosis\"\n]\n\ndef generate_yolo_labels(train_csv_path, meta_csv_path, out_dir):\n    out_dir = Path(out_dir)\n    out_dir.mkdir(parents=True, exist_ok=True)\n\n    df = pd.read_csv(train_csv_path)\n    meta_df = pd.read_csv(meta_csv_path)\n    df = df.merge(meta_df, on='image_id', how='left')\n\n    all_image_ids = df['image_id'].unique()\n\n    for img_id in all_image_ids:\n        img_df = df[df['image_id'] == img_id]\n        \n        if all(img_df['class_id'] == 14):\n            open(out_dir / f\"{img_id}.txt\", 'w').close()\n            continue\n\n        img_df = img_df[img_df['class_id'] != 14]\n        if img_df.empty:\n            open(out_dir / f\"{img_id}.txt\", 'w').close()\n            continue\n\n        orig_w = img_df['dim1'].iloc[0]\n        orig_h = img_df['dim0'].iloc[0]\n\n        boxes = []\n        labels = []\n        scores = []\n\n        for _, row in img_df.iterrows():\n            x_min = max(0.0, row['x_min'] / orig_w)\n            y_min = max(0.0, row['y_min'] / orig_h)\n            x_max = min(1.0, row['x_max'] / orig_w)\n            y_max = min(1.0, row['y_max'] / orig_h)\n\n            boxes.append([x_min, y_min, x_max, y_max])\n            labels.append(int(row['class_id']))\n            scores.append(1.0)\n\n        boxes, scores, labels = weighted_boxes_fusion([boxes], [scores], [labels], weights=None, iou_thr=0.4, skip_box_thr=0.0)\n\n        with open(out_dir / f\"{img_id}.txt\", 'w') as f:\n            for box, label in zip(boxes, labels):\n                x_min, y_min, x_max, y_max = box\n                x_center = max(0.0, min(1.0, (x_min + x_max) / 2))\n                y_center = max(0.0, min(1.0, (y_min + y_max) / 2))\n                width = max(0.0, min(1.0, x_max - x_min))\n                height = max(0.0, min(1.0, y_max - y_min))\n                f.write(f\"{int(label)} {x_center:.6f} {y_center:.6f} {width:.6f} {height:.6f}\\n\")\n\ndef build_oversampled_dataset(labels_dir, images_dir, out_dir, no_finding_fraction=0.15):\n    labels_dir = Path(labels_dir)\n    images_dir = Path(images_dir)\n    out_img = Path(out_dir) / \"images\"\n    out_lbl = Path(out_dir) / \"labels\"\n    out_img.mkdir(parents=True, exist_ok=True)\n    out_lbl.mkdir(parents=True, exist_ok=True)\n\n    class_to_imgs = defaultdict(list)\n    empty_imgs = []\n\n    for txt in labels_dir.glob(\"*.txt\"):\n        lines = txt.read_text().strip().split(\"\\n\")\n        lines = [l for l in lines if l.strip()]\n        if not lines:\n            empty_imgs.append(txt)\n            continue\n        classes = [int(l.split()[0]) for l in lines]\n        primary = max(set(classes), key=classes.count)\n        class_to_imgs[primary].append(txt)\n\n    copied = set()\n    copy_count = 0\n\n    def _copy(txt_path, suffix=\"\"):\n        nonlocal copy_count\n        img_stem = txt_path.stem\n        for ext in [\".jpg\", \".png\", \".jpeg\"]:\n            img_src = images_dir / f\"{img_stem}{ext}\"\n            if img_src.exists():\n                dst_name = f\"{img_stem}{suffix}\"\n                dst_img = out_img / f\"{dst_name}{ext}\"\n                dst_lbl = out_lbl / f\"{dst_name}.txt\"\n                if dst_img.name not in copied:\n                    shutil.copy2(img_src, dst_img)\n                    shutil.copy2(txt_path, dst_lbl)\n                    copied.add(dst_img.name)\n                    copy_count += 1\n                return True\n        return False\n\n    for cls_id, img_list in class_to_imgs.items():\n        weight = CLASS_SAMPLE_WEIGHTS.get(cls_id, 1.0)\n        repeats = max(1, round(weight))\n        for txt_path in img_list:\n            for r in range(repeats):\n                suffix = f\"_rep{r}\" if r > 0 else \"\"\n                _copy(txt_path, suffix)\n\n    n_negatives = int(len(empty_imgs) * no_finding_fraction)\n    random.shuffle(empty_imgs)\n    for txt_path in empty_imgs[:n_negatives]:\n        _copy(txt_path)\n\n    return copy_count\n\ndef tile_image_and_label(img_path, lbl_path, tile_size=512, overlap=64, out_img_dir=None, out_lbl_dir=None):\n    img = Image.open(img_path)\n    W, H = img.size\n    stride = tile_size - overlap\n\n    lines = lbl_path.read_text().strip().split(\"\\n\") if lbl_path.exists() else []\n    lines = [l for l in lines if l.strip()]\n\n    tile_idx = 0\n    for y0 in range(0, H - overlap, stride):\n        for x0 in range(0, W - overlap, stride):\n            x1 = min(x0 + tile_size, W)\n            y1 = min(y0 + tile_size, H)\n            tile_w = x1 - x0\n            tile_h = y1 - y0\n\n            tile_boxes = []\n            for line in lines:\n                parts = line.split()\n                cls = int(parts[0])\n                cx, cy, bw, bh = map(float, parts[1:5])\n                px_cx = cx * W\n                px_cy = cy * H\n                px_bw = bw * W\n                px_bh = bh * H\n                px_x1 = px_cx - px_bw / 2\n                px_y1 = px_cy - px_bh / 2\n                px_x2 = px_cx + px_bw / 2\n                px_y2 = px_cy + px_bh / 2\n\n                ix1 = max(px_x1, x0)\n                iy1 = max(px_y1, y0)\n                ix2 = min(px_x2, x1)\n                iy2 = min(px_y2, y1)\n\n                if ix2 <= ix1 or iy2 <= iy1:\n                    continue\n\n                inter_area = (ix2 - ix1) * (iy2 - iy1)\n                box_area = px_bw * px_bh\n                if box_area == 0 or inter_area / box_area < 0.50:\n                    continue\n\n                ncx = ((ix1 + ix2) / 2 - x0) / tile_w\n                ncy = ((iy1 + iy2) / 2 - y0) / tile_h\n                nbw = (ix2 - ix1) / tile_w\n                nbh = (iy2 - iy1) / tile_h\n                ncx = max(0, min(1, ncx))\n                ncy = max(0, min(1, ncy))\n                nbw = max(0, min(1, nbw))\n                nbh = max(0, min(1, nbh))\n                tile_boxes.append(f\"{cls} {ncx:.6f} {ncy:.6f} {nbw:.6f} {nbh:.6f}\")\n\n            if not tile_boxes:\n                tile_idx += 1\n                continue\n\n            tile_img = img.crop((x0, y0, x1, y1))\n            stem = img_path.stem\n            tile_name = f\"{stem}_tile{tile_idx}\"\n            tile_img.save(out_img_dir / f\"{tile_name}.jpg\", quality=95)\n            with open(out_lbl_dir / f\"{tile_name}.txt\", \"w\") as f:\n                f.write(\"\\n\".join(tile_boxes))\n            tile_idx += 1\n\n    return tile_idx\n\ndef apply_tiling_for_small_classes(base_dataset_dir, small_classes=[2, 8], tile_size=512):\n    img_dir = Path(base_dataset_dir) / \"images\"\n    lbl_dir = Path(base_dataset_dir) / \"labels\"\n    out_img = Path(base_dataset_dir) / \"images\"\n    out_lbl = Path(base_dataset_dir) / \"labels\"\n\n    tiled = 0\n    for txt in lbl_dir.glob(\"*.txt\"):\n        lines = txt.read_text().strip().split(\"\\n\")\n        has_small = any(int(l.split()[0]) in small_classes for l in lines if l.strip())\n        if not has_small:\n            continue\n        img_path = None\n        for ext in [\".jpg\", \".png\"]:\n            p = img_dir / f\"{txt.stem}{ext}\"\n            if p.exists():\n                img_path = p\n                break\n        if img_path is None:\n            continue\n        n = tile_image_and_label(img_path, txt, tile_size=tile_size, overlap=64, out_img_dir=out_img, out_lbl_dir=out_lbl)\n        tiled += n\n\n    return tiled\n\ndef write_dataset_yaml(train_dir, val_dir, out_path):\n    data = {\n        \"train\": str(train_dir),\n        \"val\": str(val_dir),\n        \"nc\": 14,\n        \"names\": CLASS_NAMES_14,\n    }\n    with open(out_path, \"w\") as f:\n        yaml.dump(data, f, allow_unicode=True, default_flow_style=False)\n\ndef prepare_data(src_images_dir, train_csv_path, meta_csv_path, output_base_dir, n_folds=5, fold_to_prepare=0):\n    temp_labels_dir = Path(output_base_dir) / \"temp_labels\"\n    generate_yolo_labels(train_csv_path, meta_csv_path, temp_labels_dir)\n\n    src_images = Path(src_images_dir)\n    src_labels = temp_labels_dir\n\n    image_ids, primary_classes = [], []\n    for txt in src_labels.glob(\"*.txt\"):\n        lines = [l for l in txt.read_text().strip().split(\"\\n\") if l.strip()]\n        if not lines:\n            continue\n        classes = [int(l.split()[0]) for l in lines]\n        image_ids.append(txt.stem)\n        primary_classes.append(max(set(classes), key=classes.count))\n\n    sgkf = StratifiedGroupKFold(n_splits=n_folds, shuffle=True, random_state=42)\n    fold_assignment = {}\n    for fold, (_, val_idx) in enumerate(sgkf.split(image_ids, primary_classes, groups=image_ids)):\n        for i in val_idx:\n            fold_assignment[image_ids[i]] = fold\n\n    train_ids = [iid for iid, f in fold_assignment.items() if f != fold_to_prepare]\n    val_ids   = [iid for iid, f in fold_assignment.items() if f == fold_to_prepare]\n\n    train_dir = Path(output_base_dir) / f\"fold{fold_to_prepare}\" / \"train\"\n    val_dir   = Path(output_base_dir) / f\"fold{fold_to_prepare}\" / \"val\"\n\n    build_oversampled_dataset(str(src_labels), str(src_images), str(train_dir), 0.15)\n    apply_tiling_for_small_classes(str(train_dir), [2, 8], 512)\n\n    val_img_dir = val_dir / \"images\"\n    val_lbl_dir = val_dir / \"labels\"\n    val_img_dir.mkdir(parents=True, exist_ok=True)\n    val_lbl_dir.mkdir(parents=True, exist_ok=True)\n\n    for iid in val_ids:\n        for ext in [\".jpg\", \".png\"]:\n            p = src_images / f\"{iid}{ext}\"\n            if p.exists():\n                shutil.copy2(p, val_img_dir / p.name)\n                break\n        lbl = src_labels / f\"{iid}.txt\"\n        if lbl.exists():\n            shutil.copy2(lbl, val_lbl_dir / lbl.name)\n\n    yaml_path = Path(output_base_dir) / f\"fold{fold_to_prepare}\" / \"dataset.yaml\"\n    write_dataset_yaml(str(train_dir / \"images\"), str(val_img_dir), str(yaml_path))\n\nif __name__ == \"__main__\":\n    prepare_data(\n        src_images_dir=\"/kaggle/input/datasets/xhlulu/vinbigdata-chest-xray-resized-png-1024x1024/train\",\n        train_csv_path=\"/kaggle/input/competitions/vinbigdata-chest-xray-abnormalities-detection/train.csv\",\n        meta_csv_path=\"/kaggle/input/datasets/xhlulu/vinbigdata-chest-xray-resized-png-1024x1024/train_meta.csv\",\n        output_base_dir=\"/kaggle/working/processed_dataset\",\n        n_folds=5,\n        fold_to_prepare=0\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T14:52:22.968311Z","iopub.execute_input":"2026-05-10T14:52:22.969491Z","iopub.status.idle":"2026-05-10T14:56:48.294238Z","shell.execute_reply.started":"2026-05-10T14:52:22.969454Z","shell.execute_reply":"2026-05-10T14:56:48.29276Z"}},"outputs":[],"execution_count":null}]}