{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13851420,"sourceType":"competition"},{"sourceId":12925487,"sourceType":"datasetVersion","datasetId":8178911},{"sourceId":12934565,"sourceType":"datasetVersion","datasetId":8184998},{"sourceId":12964800,"sourceType":"datasetVersion","datasetId":8198385,"isSourceIdPinned":true},{"sourceId":12973967,"sourceType":"datasetVersion","datasetId":8211571},{"sourceId":12974708,"sourceType":"datasetVersion","datasetId":8212035},{"sourceId":13081495,"sourceType":"datasetVersion","datasetId":8278043},{"sourceId":13222199,"sourceType":"datasetVersion","datasetId":7979194},{"sourceId":226368929,"sourceType":"kernelVersion"},{"sourceId":259333705,"sourceType":"kernelVersion"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install ultralytics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T07:48:46.849308Z","iopub.execute_input":"2025-09-17T07:48:46.849676Z","iopub.status.idle":"2025-09-17T07:48:53.7056Z","shell.execute_reply.started":"2025-09-17T07:48:46.849648Z","shell.execute_reply":"2025-09-17T07:48:53.704465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, glob, math, random\nimport ast\nfrom pathlib import Path\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\nfrom PIL import Image, ImageDraw, ImageFont\nimport pandas as pd\nimport cv2\nimport torch\nimport pydicom\nfrom scipy.ndimage import zoom\nfrom ultralytics import YOLO","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T07:48:53.707123Z","iopub.execute_input":"2025-09-17T07:48:53.707452Z","iopub.status.idle":"2025-09-17T07:48:59.427297Z","shell.execute_reply.started":"2025-09-17T07:48:53.707414Z","shell.execute_reply":"2025-09-17T07:48:59.426228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ========== 小物ユーティリティ ==========\ndef _iou_xyxy(a, b):\n    # a,b: [x1,y1,x2,y2]\n    x1 = max(a[0], b[0]); y1 = max(a[1], b[1])\n    x2 = min(a[2], b[2]); y2 = min(a[3], b[3])\n    iw = max(0.0, x2 - x1); ih = max(0.0, y2 - y1)\n    inter = iw * ih\n    if inter <= 0: return 0.0\n    area_a = max(0.0, (a[2]-a[0]) * (a[3]-a[1]))\n    area_b = max(0.0, (b[2]-b[0]) * (b[3]-b[1]))\n    union = area_a + area_b - inter + 1e-9\n    return inter / union\n\ndef _ensure_dir(p):\n    Path(p).mkdir(parents=True, exist_ok=True)\n\n# ========== 簡易 Weighted Boxes Fusion（クラス別） ==========\ndef weighted_boxes_fusion(\n    boxes, scores, labels,\n    iou_thr=0.55,\n    conf_type=\"avg\"   # \"avg\" か \"one_minus_prod\"\n):\n    \"\"\"\n    複数モデルの出力をマージ。boxes/scores/labels はすべて結合済み1配列（同一画像）。\n    - boxes: (N,4) xyxy\n    - scores: (N,)\n    - labels: (N,)\n    返り値: fused_boxes, fused_scores, fused_labels\n    \"\"\"\n    if len(boxes) == 0:\n        return boxes, scores, labels\n\n    order = np.argsort(-scores)\n    boxes = boxes[order]; scores = scores[order]; labels = labels[order]\n\n    clusters = []  # list of dict: {'box': np.array(4), 'score_sum': float, 'scores': [..], 'label': int, 'weight_sum': float, 'members': int}\n    for b, s, c in zip(boxes, scores, labels):\n        matched = False\n        for cl in clusters:\n            if cl['label'] != c:\n                continue\n            if _iou_xyxy(b, cl['box']) >= iou_thr:\n                # 既存クラスタにマージ（座標はスコア重みで移動平均）\n                w_old = cl['weight_sum']; w_new = s\n                cl['box'] = (cl['box'] * w_old + b * w_new) / (w_old + w_new + 1e-9)\n                cl['weight_sum'] += w_new\n                cl['scores'].append(s)\n                cl['members'] += 1\n                matched = True\n                break\n        if not matched:\n            clusters.append({\n                'box': b.copy(),\n                'label': int(c),\n                'weight_sum': float(s),\n                'scores': [float(s)],\n                'members': 1,\n            })\n\n    fused_boxes, fused_scores, fused_labels = [], [], []\n    for cl in clusters:\n        fused_boxes.append(cl['box'])\n        if conf_type == \"one_minus_prod\":\n            # 1 - ∏(1-s) で統合信頼度\n            p = 1.0\n            for s in cl['scores']:\n                p *= (1.0 - s)\n            sc = 1.0 - p\n        else:\n            sc = float(np.mean(cl['scores']))\n        fused_scores.append(sc)\n        fused_labels.append(cl['label'])\n\n    return np.vstack(fused_boxes), np.array(fused_scores), np.array(fused_labels, dtype=int)\n\n# ========== 1画像に対する fold アンサンブル推論 ==========\ndef predict_ensemble_single_image(\n    models,\n    image,\n    conf=0.2,\n    iou=0.7,\n    imgsz=320,\n    per_model_max_det=1,      # 各モデルが返す最大検出数\n    ensemble_iou=0.55,          # WBF のクラスタ閾値\n    only_one=True,             # True で最終1件に絞る\n    agnostic=False,             # True でクラス無視の後段NMSをしたい場合（今回はWBFなので通常Falseのまま推奨）\n):\n    \"\"\"\n    各 fold の best.pt を読み、単一画像に対して推論→WBFで統合。\n    戻り値: dict {'boxes': (K,4), 'scores': (K,), 'labels': (K,)}\n    \"\"\"\n    # 遅延ロードを避けたい場合は、外で [YOLO(p) for p in model_paths] を渡す実装に変えてもOK\n    boxes_all = []\n    scores_all = []\n    labels_all = []\n\n    for model in models:\n        model = YOLO(mp)\n        res = model.predict(\n            source=image,\n            conf=conf, iou=iou, imgsz=imgsz, max_det=per_model_max_det, verbose=False\n        )[0]\n        if res.boxes is None or len(res.boxes) == 0:\n            continue\n        xyxy = res.boxes.xyxy.cpu().numpy()\n        sco  = res.boxes.conf.cpu().numpy()\n        lab  = (res.boxes.cls.cpu().numpy().astype(int)\n                if res.boxes.cls is not None else np.zeros(len(xyxy), dtype=int))\n        boxes_all.append(xyxy); scores_all.append(sco); labels_all.append(lab)\n\n    if len(boxes_all) == 0:\n        return {'boxes': np.zeros((0,4), dtype=np.float32),\n                'scores': np.zeros((0,), dtype=np.float32),\n                'labels': np.zeros((0,), dtype=np.int32)}\n\n    boxes = np.vstack(boxes_all)\n    scores = np.hstack(scores_all)\n    labels = np.hstack(labels_all)\n\n    # 簡易 WBF\n    f_boxes, f_scores, f_labels = weighted_boxes_fusion(\n        boxes, scores, labels, iou_thr=ensemble_iou, conf_type=\"one_minus_prod\"\n    )\n\n    # 1件に絞る（スコア最大）\n    if only_one and len(f_scores) > 0:\n        idx = int(np.argmax(f_scores))\n        f_boxes = f_boxes[idx:idx+1]\n        f_scores = f_scores[idx:idx+1]\n        f_labels = f_labels[idx:idx+1]\n\n    return {'boxes': f_boxes, 'scores': f_scores, 'labels': f_labels}\n\n# ========== ディレクトリ一括推論 & オーバーレイ保存 & CSV ==========\ndef ensemble_infer_dir(\n    model_paths,\n    images_root,\n    out_dir=\"/kaggle/working/ens_out\",\n    conf=0.2,\n    iou=0.7,\n    imgsz=640,\n    per_model_max_det=100,\n    ensemble_iou=0.55,\n    only_one=False,\n    draw_score=True,\n):\n    \"\"\"\n    images_root 配下の *.png / *.jpg を再帰探索して、fold アンサンブル推論。\n    - オーバーレイ画像を out_dir/overlays 下に保存（元のサブフォルダ構造を維持）\n    - 予測座標を CSV に保存（out_dir/ensemble_predictions.csv）\n    \"\"\"\n    images_root = Path(images_root)\n    out_dir = Path(out_dir)\n    overlays_dir = out_dir / \"overlays\"\n    _ensure_dir(overlays_dir)\n\n    img_paths = (list(images_root.rglob(\"*.png\")) +\n                 list(images_root.rglob(\"*.jpg\")) +\n                 list(images_root.rglob(\"*.jpeg\")))\n    img_paths.sort()\n    if not img_paths:\n        print(f\"No images found under: {images_root}\")\n        return None\n\n    rows = []\n    # フォント（環境によっては使えないことがあるので None でフォールバック）\n    try:\n        font = ImageFont.truetype(\"DejaVuSans.ttf\", size=14)\n    except:\n        font = None\n\n    for img_path in img_paths:\n        pred = predict_ensemble_single_image(\n            model_paths, str(img_path),\n            conf=conf, iou=iou, imgsz=imgsz,\n            per_model_max_det=per_model_max_det,\n            ensemble_iou=ensemble_iou,\n            only_one=only_one,\n        )\n\n        # CSV 行を作成\n        if len(pred['scores']) == 0:\n            rows.append({\n                \"image_path\": str(img_path),\n                \"x1\": np.nan, \"y1\": np.nan, \"x2\": np.nan, \"y2\": np.nan,\n                \"score\": np.nan, \"class\": np.nan,\n                \"cx\": np.nan, \"cy\": np.nan, \"w\": np.nan, \"h\": np.nan\n            })\n        else:\n            for (x1,y1,x2,y2), sc, cls in zip(pred['boxes'], pred['scores'], pred['labels']):\n                w = x2 - x1; h = y2 - y1; cx = x1 + w/2; cy = y1 + h/2\n                rows.append({\n                    \"image_path\": str(img_path),\n                    \"x1\": float(x1), \"y1\": float(y1), \"x2\": float(x2), \"y2\": float(y2),\n                    \"score\": float(sc), \"class\": int(cls),\n                    \"cx\": float(cx), \"cy\": float(cy), \"w\": float(w), \"h\": float(h)\n                })\n\n        # オーバーレイ保存\n        img = Image.open(img_path).convert(\"RGB\")\n        draw = ImageDraw.Draw(img)\n        for (x1,y1,x2,y2), sc, cls in zip(pred['boxes'], pred['scores'], pred['labels']):\n            draw.rectangle([x1,y1,x2,y2], outline=(255,0,0), width=2)\n            if draw_score:\n                txt = f\"{int(cls)}:{sc:.2f}\"\n                tw, th = draw.textlength(txt, font=font) if hasattr(draw, \"textlength\") else (len(txt)*8, 12)\n                draw.rectangle([x1, max(0,y1-th-2), x1+tw+4, y1], fill=(255,0,0))\n                draw.text((x1+2, max(0,y1-th-2)), txt, fill=(255,255,255), font=font)\n\n        rel = img_path.relative_to(images_root)\n        out_img_path = overlays_dir / rel\n        _ensure_dir(out_img_path.parent)\n        img.save(out_img_path)\n\n    # CSV 保存\n    df = pd.DataFrame(rows)\n    csv_path = out_dir / \"ensemble_predictions.csv\"\n    _ensure_dir(csv_path.parent)\n    df.to_csv(csv_path, index=False)\n    print(f\"Saved overlays to: {overlays_dir}\")\n    print(f\"Saved CSV to:      {csv_path}\")\n    return {\"csv\": str(csv_path), \"overlays_dir\": str(overlays_dir), \"df\": df}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T07:48:59.429489Z","iopub.execute_input":"2025-09-17T07:48:59.430084Z","iopub.status.idle":"2025-09-17T07:48:59.458489Z","shell.execute_reply.started":"2025-09-17T07:48:59.430045Z","shell.execute_reply":"2025-09-17T07:48:59.457258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pct_normalize(img, p_min, p_max):\n    vmin = np.percentile(img, p_min)\n    vmax = np.percentile(img, p_max)\n    img = np.clip(img, vmin, vmax)\n    img = (img - vmin) / (vmax - vmin + 1e-6)\n    img = (img*255).astype(np.uint8)\n    return img\n\ndef value_normalize(img, vmin, vmax):\n    img = np.clip(img, vmin, vmax)\n    img = (img - vmin) / (vmax - vmin + 1e-6)\n    img = (img*255).astype(np.uint8)\n    return img\n    \ndef resample_zyx(arr: np.ndarray,\n                 spacing: tuple[float, float, float],\n                 new_spacing: tuple[float, float, float],\n                 order: int) -> np.ndarray:\n    \"\"\"\n    (Z,Y,X) を spacing -> new_spacing へリサンプリング。\n    order=1: 画像（線形）, order=0: マスク（最近傍）\n    \"\"\"\n    sz, sy, sx = spacing\n    nsz, nsy, nsx = new_spacing\n    zz, zy, zx = sz / nsz, sy / nsy, sx / nsx\n    mode = \"nearest\" if order == 0 else \"constant\"\n    return zoom(arr, zoom=(zz, zy, zx), order=order, mode=mode)\n    \ndef load_dicom_array_2d(paths):\n    imgs = []\n    for path in paths:\n        dcm = pydicom.dcmread(path)\n        imgs.append(dcm.pixel_array)\n    imgs = np.array(imgs)\n    return imgs\n\ndef load_dicom_array_3d(path):\n    dcm = pydicom.dcmread(path)\n    imgs = dcm.pixel_array\n    return imgs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T07:48:59.460138Z","iopub.execute_input":"2025-09-17T07:48:59.46057Z","iopub.status.idle":"2025-09-17T07:48:59.482753Z","shell.execute_reply.started":"2025-09-17T07:48:59.460539Z","shell.execute_reply":"2025-09-17T07:48:59.481371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def bbox_coords(row):\n    df_sample = df_bbox[df_bbox['SeriesInstanceUID'] == row['SeriesInstanceUID']]\n    zs_min, zs_max, ys_min, ys_max, xs_min, xs_max = [], [], [], [], [], []\n    for _, row_bbox in df_sample.iterrows():\n        x1, x2, y1, y2 = row_bbox['x1'], row_bbox['x2'], row_bbox['y1'], row_bbox['y2']\n        if row['OrientationLabel']=='AXIAL':\n            if row_bbox['axis']=='axis2':\n                zs_min.append(y1)\n                zs_max.append(y2)\n                ys_min.append(x1)\n                ys_max.append(x2)\n            elif row_bbox['axis']=='axis1':\n                zs_min.append(y1)\n                zs_max.append(y2)\n                xs_min.append(x1)\n                xs_max.append(x2)\n        elif row['OrientationLabel']=='SAGITTAL':\n            if row_bbox['axis']=='axis2':\n                zs_min.append(y1)\n                zs_max.append(y2)\n                ys_min.append(x1)\n                ys_max.append(x2)\n            elif row_bbox['axis']=='axis0':\n                ys_min.append(y1)\n                ys_max.append(y2)\n                xs_min.append(x1)\n                xs_max.append(x2)\n        elif row['OrientationLabel']=='CORONAL':\n            if row_bbox['axis']=='axis2':\n                zs_min.append(y1)\n                zs_max.append(y2)\n                ys_min.append(x1)\n                ys_max.append(x2)\n            elif row_bbox['axis']=='axis0':\n                ys_min.append(y1)\n                ys_max.append(y2)\n                xs_min.append(x1)\n                xs_max.append(x2)\n    z_min_mean = np.nanmean(zs_min)\n    z_max_mean = np.nanmean(zs_max)\n    y_min_mean = np.nanmean(ys_min)\n    y_max_mean = np.nanmean(ys_max)\n    x_min_mean = np.nanmean(xs_min)\n    x_max_mean = np.nanmean(xs_max)\n    return pd.Series({'x1_mm': x_min_mean,\n                      'x2_mm': x_max_mean,\n                      'y1_mm': y_min_mean,\n                      'y2_mm': y_max_mean,\n                      'z1_mm': z_min_mean,\n                      'z2_mm': z_max_mean\n                     })\n\n\nimport numpy as np\nimport pandas as pd\nfrom collections.abc import Sequence\n\ndef restore_coordinates_zyx(row) -> pd.Series:\n    \"\"\"\n    (Z, Y, X) を spacing(=1,1,1) -> new_spacing へスケール変換。\n    何らかのエラー（欠損/型不正/0除算など）の場合は NaN を返す。\n    期待キー: 'z_spacing', 'PixelSpacing', 'x1','x2','y1','y2','z1','z2'\n    \"\"\"\n    keys = ['x1', 'x2', 'y1', 'y2', 'z1', 'z2']\n    def na_series():\n        return pd.Series({k: np.nan for k in keys})\n\n    try:\n        # 元の仮定: 旧座標は spacing=(1,1,1) ベース\n        sz = sy = sx = 1.0\n\n        # 目標の新しい spacing を取り出し\n        nsz = float(row['z_spacing'])\n\n        ps = row['PixelSpacing']\n        # PixelSpacing の頑健なパース（list/tuple/pydicom MultiValue/str）\n        if isinstance(ps, str):\n            s = ps.strip().strip('[]()')\n            parts = [p for p in s.replace(';', ',').split(',') if p.strip() != '']\n            if len(parts) < 2:\n                return na_series()\n            nsy = float(parts[0])\n            nsx = float(parts[1])\n        elif isinstance(ps, Sequence):\n            if len(ps) < 2:\n                return na_series()\n            nsy = float(ps[0])\n            nsx = float(ps[1])\n        else:\n            return na_series()\n\n        # 0 または 非有限値は無効\n        if not np.isfinite(nsz) or not np.isfinite(nsy) or not np.isfinite(nsx):\n            return na_series()\n        if nsz == 0.0 or nsy == 0.0 or nsx == 0.0:\n            return na_series()\n\n        # スケール係数\n        zz, zy, zx = sz / nsz, sy / nsy, sx / nsx\n\n        vals = {\n            'x1': float(row['x1_mm']) * zx,\n            'x2': float(row['x2_mm']) * zx,\n            'y1': float(row['y1_mm']) * zy,\n            'y2': float(row['y2_mm']) * zy,\n            'z1': float(row['z1_mm']) * zz,\n            'z2': float(row['z2_mm']) * zz,\n        }\n        # 数値でない場合は except に落ちる\n        return pd.Series(vals)\n\n    except Exception:\n        return na_series()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T07:48:59.484541Z","iopub.execute_input":"2025-09-17T07:48:59.484869Z","iopub.status.idle":"2025-09-17T07:48:59.503219Z","shell.execute_reply.started":"2025-09-17T07:48:59.484832Z","shell.execute_reply":"2025-09-17T07:48:59.502305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"root_pred = '/kaggle/input/rsna2025-series-yolo-prediction-plus3d/ens_out/overlays/'\ndf = pd.read_csv('/kaggle/input/rsna2025-extra/train_add_metadata_v5.csv')\ndf[\"sorted_files\"] = df[\"sorted_files\"].apply(lambda x: ast.literal_eval(x) if pd.notna(x) else x)\ndf['PixelSpacing'] = df['PixelSpacing'].apply(lambda x: ast.literal_eval(x) if pd.notna(x) else x)\n#df_bbox = pd.read_csv('/kaggle/input/rsna2025-series-yolo-prediction-plus3d/ens_out/ensemble_predictions.csv')\ndf_bbox = pd.read_csv('/kaggle/input/rsna2025-series-yolo-prediction-plus3d/ens_out/ensemble_predictions.csv')\ndf_bbox['SeriesInstanceUID'] = df_bbox['image_path'].apply(lambda x: x.split('/')[-1][:-4])\ndf_bbox['image_path'] = df_bbox['image_path'].apply(lambda x: root_pred+'/'.join(x.split('/')[-5:]))\ndf_bbox['axis'] = df_bbox['image_path'].apply(lambda x: x.split('/')[-3])\n\n# img_files = row['sorted_files']\n# sid = row['SeriesInstanceUID']\n# modality = row['Modality']\n# plane = row['OrientationLabel']\n# if np.isnan(img_files):\n#     img_file = list(Path(f'/kaggle/input/rsna-intracranial-aneurysm-detection/series/{sid}').glob('*'))[0]\n#     imgs = load_dicom_array_3d(img_file)\n# else:\n#     imgs = load_dicom_array_2d(img_files)\n#     y_spacing, x_spacing = row['PixelSpacing']\n#     z_spacing = row['z_spacing']\n#     imgs = resample_zyx(imgs,\n#                         spacing=(z_spacing, y_spacing, x_spacing),\n#                         new_spacing=(1., 1., 1.),\n#                         order=1\n#                        )\n# if modality=='CTA':\n#     imgs = value_normalize(imgs, -100, 600)\n# else:\n#     imgs = pct_normalize(imgs, 1, 99)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T07:48:59.504203Z","iopub.execute_input":"2025-09-17T07:48:59.504544Z","iopub.status.idle":"2025-09-17T07:49:06.820683Z","shell.execute_reply.started":"2025-09-17T07:48:59.504517Z","shell.execute_reply":"2025-09-17T07:49:06.81993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T07:49:06.82182Z","iopub.execute_input":"2025-09-17T07:49:06.822165Z","iopub.status.idle":"2025-09-17T07:49:06.829173Z","shell.execute_reply.started":"2025-09-17T07:49:06.822132Z","shell.execute_reply":"2025-09-17T07:49:06.828017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[['x1_mm', 'x2_mm', 'y1_mm', 'y2_mm', 'z1_mm', 'z2_mm']] = df.apply(bbox_coords, axis=1)\ndf[['x1', 'x2', 'y1', 'y2', 'z1', 'z2']] = df.apply(restore_coordinates_zyx, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T07:49:08.86497Z","iopub.execute_input":"2025-09-17T07:49:08.865376Z","iopub.status.idle":"2025-09-17T07:49:18.392671Z","shell.execute_reply.started":"2025-09-17T07:49:08.865337Z","shell.execute_reply":"2025-09-17T07:49:18.39178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['x_size'] = df['x2_mm'] - df['x1_mm']\ndf['y_size'] = df['y2_mm'] - df['y1_mm']\ndf['z_size'] = df['z2_mm'] - df['z1_mm']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T07:49:18.393464Z","iopub.execute_input":"2025-09-17T07:49:18.393702Z","iopub.status.idle":"2025-09-17T07:49:18.400152Z","shell.execute_reply.started":"2025-09-17T07:49:18.393681Z","shell.execute_reply":"2025-09-17T07:49:18.399229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure()\nfor plane in ['AXIAL', 'SAGITTAL', 'CORONAL']:\n    df_plane = df[df['OrientationLabel']==plane]\n    df_plane['x_size'].hist(label=plane, alpha=0.3, density=True)\nplt.title('x_size')\nplt.xlabel('mm')\nplt.legend()\n\nplt.figure()\nfor plane in ['AXIAL', 'SAGITTAL', 'CORONAL']:\n    df_plane = df[df['OrientationLabel']==plane]\n    df_plane['y_size'].hist(label=plane, alpha=0.3, density=True)\nplt.title('y_size')\nplt.xlabel('mm')\nplt.legend()\n\nplt.figure()\nfor plane in ['AXIAL', 'SAGITTAL', 'CORONAL']:\n    df_plane = df[df['OrientationLabel']==plane]\n    df_plane['z_size'].hist(label=plane, alpha=0.3, density=True)\nplt.title('z_size')\nplt.xlabel('mm')\nplt.legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T07:49:18.401107Z","iopub.execute_input":"2025-09-17T07:49:18.401491Z","iopub.status.idle":"2025-09-17T07:49:19.544445Z","shell.execute_reply.started":"2025-09-17T07:49:18.401453Z","shell.execute_reply":"2025-09-17T07:49:19.543488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.to_csv('train_add_metadata_v5_roi.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-17T07:49:19.546095Z","iopub.execute_input":"2025-09-17T07:49:19.546457Z","iopub.status.idle":"2025-09-17T07:49:24.731358Z","shell.execute_reply.started":"2025-09-17T07:49:19.546428Z","shell.execute_reply":"2025-09-17T07:49:24.730482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# idx = 1\n# df_plane = df_seg[df_seg['OrientationLabel']=='CORONAL']\n# sid = df_plane['SeriesInstanceUID'].iloc[idx]\n# row = df_seg[df_seg['SeriesInstanceUID']==sid].iloc[0]\n# img_files = row['sorted_files']\n# sid = row['SeriesInstanceUID']\n# modality = row['Modality']\n# plane = row['OrientationLabel']\n# if not isinstance(img_files, list):\n#     img_file = list(Path(f'/kaggle/input/rsna-intracranial-aneurysm-detection/series/{sid}').glob('*'))[0]\n#     imgs = load_dicom_array_3d(img_file)\n# else:\n#     imgs = load_dicom_array_2d(img_files)\n#     y_spacing, x_spacing = row['PixelSpacing']\n#     z_spacing = row['z_spacing']\n#     # imgs = resample_zyx(imgs,\n#     #                     spacing=(z_spacing, y_spacing, x_spacing),\n#     #                     new_spacing=(1., 1., 1.),\n#     #                     order=1\n#     #                    )\n# if modality=='CTA':\n#     imgs = value_normalize(imgs, -100, 600)\n# else:\n#     imgs = pct_normalize(imgs, 1, 99)\n# try:\n#     x1, x2, y1, y2, z1, z2 = int(row['x1']), int(row['x2']), int(row['y1']), int(row['y2']), int(row['z1']), int(row['z2'])\n# except:\n#     x1, x2, y1, y2, z1, z2 = 0, 0, 0, 0, 0, 0\n# Z, H, W = imgs.shape\n# z_mid, h_mid, w_mid = Z//2, H//2, W//2\n# img = imgs[z_mid]\n# cv2.rectangle(img, (x1, y1), (x2, y2), 255, 2)\n# plt.figure(figsize=(5,5))\n# plt.imshow(img)\n# img = imgs[:,h_mid,:]\n# cv2.rectangle(img, (x1, z1), (x2, z2), 255, 2)\n# plt.figure(figsize=(5,5))\n# plt.imshow(img)\n# img = np.ascontiguousarray(imgs[:,:,w_mid])\n# cv2.rectangle(img, (y1, z1), (y2, z2), 255, 2)\n# plt.figure(figsize=(5,5))\n# plt.imshow(img)\n\n# files = Path('/kaggle/input/rsna2025-series-yolo-prediction/ens_out/overlays').glob(f'*/*/*/images/{sid}*')\n# for file in files:\n#     img = cv2.imread(file)\n#     plt.figure(figsize=(5,5))\n#     plt.imshow(img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T23:17:35.122341Z","iopub.execute_input":"2025-09-16T23:17:35.122638Z","iopub.status.idle":"2025-09-16T23:17:38.337659Z","shell.execute_reply.started":"2025-09-16T23:17:35.122611Z","shell.execute_reply":"2025-09-16T23:17:38.336585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df_miss = df_seg.loc[df_seg['x1'].isna()&df_seg['y1'].isna()&df_seg['z1'].isna()].reset_index(drop=True)\n# df_miss","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-16T23:17:38.371233Z","iopub.execute_input":"2025-09-16T23:17:38.371531Z","iopub.status.idle":"2025-09-16T23:17:38.421108Z","shell.execute_reply.started":"2025-09-16T23:17:38.371503Z","shell.execute_reply":"2025-09-16T23:17:38.419951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}