{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13851420,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **00 - Set Up**","metadata":{}},{"cell_type":"code","source":"%pip install celluloid --q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:54:54.467133Z","iopub.execute_input":"2025-11-05T09:54:54.467506Z","iopub.status.idle":"2025-11-05T09:54:59.488696Z","shell.execute_reply.started":"2025-11-05T09:54:54.467479Z","shell.execute_reply":"2025-11-05T09:54:59.486317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%matplotlib notebook\n    \nfrom pathlib import Path\nimport nibabel as nib\nimport pandas as pd\nimport numpy as np\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom celluloid import Camera\nfrom IPython.display import HTML\n\nimport seaborn as sns\nfrom matplotlib import colors\nfrom matplotlib.patches import Patch\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:54:59.491684Z","iopub.execute_input":"2025-11-05T09:54:59.492046Z","iopub.status.idle":"2025-11-05T09:55:00.664841Z","shell.execute_reply.started":"2025-11-05T09:54:59.492017Z","shell.execute_reply":"2025-11-05T09:55:00.663276Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **01 - EDA**","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv')\ndf_train_local = pd.read_csv('/kaggle/input/rsna-intracranial-aneurysm-detection/train_localizers.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:00.667228Z","iopub.execute_input":"2025-11-05T09:55:00.667932Z","iopub.status.idle":"2025-11-05T09:55:00.779562Z","shell.execute_reply.started":"2025-11-05T09:55:00.667884Z","shell.execute_reply":"2025-11-05T09:55:00.777745Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **train.csv contents**","metadata":{}},{"cell_type":"code","source":"print(df_train.shape)\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:00.782057Z","iopub.execute_input":"2025-11-05T09:55:00.784292Z","iopub.status.idle":"2025-11-05T09:55:00.836002Z","shell.execute_reply.started":"2025-11-05T09:55:00.784238Z","shell.execute_reply":"2025-11-05T09:55:00.834407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_train[\"Aneurysm Present\"].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:00.836834Z","iopub.execute_input":"2025-11-05T09:55:00.837123Z","iopub.status.idle":"2025-11-05T09:55:00.859254Z","shell.execute_reply.started":"2025-11-05T09:55:00.8371Z","shell.execute_reply":"2025-11-05T09:55:00.857667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info(verbose=True, show_counts=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:00.861187Z","iopub.execute_input":"2025-11-05T09:55:00.861691Z","iopub.status.idle":"2025-11-05T09:55:00.908065Z","shell.execute_reply.started":"2025-11-05T09:55:00.861647Z","shell.execute_reply":"2025-11-05T09:55:00.906802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:00.909537Z","iopub.execute_input":"2025-11-05T09:55:00.909921Z","iopub.status.idle":"2025-11-05T09:55:00.992853Z","shell.execute_reply.started":"2025-11-05T09:55:00.909889Z","shell.execute_reply":"2025-11-05T09:55:00.991606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:00.994073Z","iopub.execute_input":"2025-11-05T09:55:00.994479Z","iopub.status.idle":"2025-11-05T09:55:01.006278Z","shell.execute_reply.started":"2025-11-05T09:55:00.994452Z","shell.execute_reply":"2025-11-05T09:55:01.004521Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **eda plots**","metadata":{}},{"cell_type":"code","source":"features_num = ['PatientAge']\n\nfeatures_cat = ['PatientSex', 'Modality',\n       'Left Infraclinoid Internal Carotid Artery',\n       'Right Infraclinoid Internal Carotid Artery',\n       'Left Supraclinoid Internal Carotid Artery',\n       'Right Supraclinoid Internal Carotid Artery',\n       'Left Middle Cerebral Artery', 'Right Middle Cerebral Artery',\n       'Anterior Communicating Artery', 'Left Anterior Cerebral Artery',\n       'Right Anterior Cerebral Artery', 'Left Posterior Communicating Artery',\n       'Right Posterior Communicating Artery', 'Basilar Tip',\n       'Other Posterior Circulation']\n\ntarget = 'Aneurysm Present'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:01.007592Z","iopub.execute_input":"2025-11-05T09:55:01.007851Z","iopub.status.idle":"2025-11-05T09:55:01.034427Z","shell.execute_reply.started":"2025-11-05T09:55:01.00783Z","shell.execute_reply":"2025-11-05T09:55:01.032276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot histograms\nfor f in features_num:\n    plt.figure(figsize=(8,3))\n    df_train[f].plot(kind='hist', bins=25, color='darkblue')\n    plt.title(f + ' - Train')\n    plt.grid()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:01.039979Z","iopub.execute_input":"2025-11-05T09:55:01.041445Z","iopub.status.idle":"2025-11-05T09:55:01.617095Z","shell.execute_reply.started":"2025-11-05T09:55:01.041385Z","shell.execute_reply":"2025-11-05T09:55:01.615565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot categorical feature distributions\nfor f in features_cat:\n    plt.figure(figsize=(8,3))\n    df_train[f].value_counts().sort_index().plot(kind='bar', color='darkblue')\n    plt.title(f + ' - Train')\n    plt.grid()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:01.618607Z","iopub.execute_input":"2025-11-05T09:55:01.619027Z","iopub.status.idle":"2025-11-05T09:55:04.70408Z","shell.execute_reply.started":"2025-11-05T09:55:01.618992Z","shell.execute_reply":"2025-11-05T09:55:04.703024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(4,4))\ndf_train[target].value_counts().sort_index().plot(kind='bar', color='darkblue')\nplt.title(target)\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:04.705448Z","iopub.execute_input":"2025-11-05T09:55:04.705763Z","iopub.status.idle":"2025-11-05T09:55:04.876649Z","shell.execute_reply.started":"2025-11-05T09:55:04.705741Z","shell.execute_reply":"2025-11-05T09:55:04.875231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot features distributions split by target\nfor f in features_num:\n    plt.figure(figsize=(4,3))\n    sns.violinplot(df_train, x=target, y=f,)\n    plt.title(f + ' by target')\n    plt.grid()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:04.877955Z","iopub.execute_input":"2025-11-05T09:55:04.878527Z","iopub.status.idle":"2025-11-05T09:55:05.263974Z","shell.execute_reply.started":"2025-11-05T09:55:04.87849Z","shell.execute_reply":"2025-11-05T09:55:05.262446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# impact of categorical features - normalized cross tables\nfor f in features_cat:\n    print('>>> Feature:', f)\n    ctab = pd.crosstab(df_train[target], df_train[f])\n    ctab_norm = ctab / ctab.sum()\n    plt.figure(figsize=(5,2))\n    g = sns.heatmap(ctab_norm, annot=True,\n                    fmt='.3f', linecolor='black',\n                    linewidths=0.5, cmap='Greens', \n                    vmin=0, vmax=+1)\n    plt.title(f + ' vs target - train')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:05.265524Z","iopub.execute_input":"2025-11-05T09:55:05.265856Z","iopub.status.idle":"2025-11-05T09:55:08.17212Z","shell.execute_reply.started":"2025-11-05T09:55:05.265834Z","shell.execute_reply":"2025-11-05T09:55:08.17099Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **02 - Segmentation Visualization**","metadata":{}},{"cell_type":"code","source":"img = nib.load(\"/kaggle/input/rsna-intracranial-aneurysm-detection/segmentations/1.2.826.0.1.3680043.8.498.10035643165968342618460849823699311381.nii\")\ndata = img.get_fdata()\n\nlabels = nib.load(\"/kaggle/input/rsna-intracranial-aneurysm-detection/segmentations/1.2.826.0.1.3680043.8.498.10035643165968342618460849823699311381_cowseg.nii\")\nmask = labels.get_fdata().astype(int)\n\nprint(\"Shape:\", data.shape)\nprint(\"Unique labels:\", np.unique(mask))\n\n# Define colors: label 0 transparent, labels 1-13 distinct\ndistinct_colors = [(0,0,0,0), \"#e6194b\", \"#3cb44b\", \"#ffe119\", \"#4363d8\", \"#f58231\", \"#911eb4\",\n                   \"#46f0f0\", \"#f032e6\", \"#bcf60c\", \"#fabebe\", \"#008080\", \"#e6beff\", \"#9a6324\"]\ncmap = colors.ListedColormap(distinct_colors)\nnorm = colors.BoundaryNorm(boundaries=np.arange(-0.5, len(distinct_colors)+0.5, 1), ncolors=len(distinct_colors))\n\n# Label names\nlabel_names = {\n    1: \"Other Posterior Circulation\",\n    2: \"Basilar Tip\",\n    3: \"Right Posterior Communicating Artery\",\n    4: \"Left Posterior Communicating Artery\",\n    5: \"Right Infraclinoid Internal Carotid Artery\",\n    6: \"Left Infraclinoid Internal Carotid Artery\",\n    7: \"Right Supraclinoid Internal Carotid Artery\",\n    8: \"Left Supraclinoid Internal Carotid Artery\",\n    9: \"Right Middle Cerebral Artery\",\n    10: \"Left Middle Cerebral Artery\",\n    11: \"Right Anterior Cerebral Artery\",\n    12: \"Left Anterior Cerebral Artery\",\n    13: \"Anterior Communicating Artery\"\n}\n\n# Animation\nfig = plt.figure(figsize=(6, 6))\ncamera = Camera(fig)\n\nfor i in range(data.shape[2]):\n    plt.imshow(np.rot90(data[:, :, i]), cmap=\"gray\")\n    plt.imshow(np.rot90(mask[:, :, i]), cmap=cmap, alpha=0.7, norm=norm)\n    plt.axis('off')\n    camera.snap()\n\nanimation = camera.animate(interval=50)\n\n# Create legend for labels 1-13 (skip background 0)\nlegend_patches = [Patch(color=distinct_colors[i], label=label_names[i]) for i in label_names.keys()]\nplt.legend(handles=legend_patches, bbox_to_anchor=(1.05, 1), loc='upper left', borderaxespad=0.)\n\nHTML(animation.to_html5_video())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:55:08.173253Z","iopub.execute_input":"2025-11-05T09:55:08.173512Z","iopub.status.idle":"2025-11-05T09:56:06.828533Z","shell.execute_reply.started":"2025-11-05T09:55:08.173494Z","shell.execute_reply":"2025-11-05T09:56:06.827358Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **03 - Modality ImageShape Distribution**","metadata":{}},{"cell_type":"code","source":"%pip install torchio --q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:56:06.829808Z","iopub.execute_input":"2025-11-05T09:56:06.830101Z","iopub.status.idle":"2025-11-05T09:56:11.18242Z","shell.execute_reply.started":"2025-11-05T09:56:06.830079Z","shell.execute_reply":"2025-11-05T09:56:11.180915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\n\nimport torch\nimport torchio as tio\n\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data import Dataset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:56:11.184211Z","iopub.execute_input":"2025-11-05T09:56:11.184617Z","iopub.status.idle":"2025-11-05T09:56:11.191022Z","shell.execute_reply.started":"2025-11-05T09:56:11.184585Z","shell.execute_reply":"2025-11-05T09:56:11.189916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = pd.read_csv(\"/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:56:11.192563Z","iopub.execute_input":"2025-11-05T09:56:11.192945Z","iopub.status.idle":"2025-11-05T09:56:11.238996Z","shell.execute_reply.started":"2025-11-05T09:56:11.192913Z","shell.execute_reply":"2025-11-05T09:56:11.23794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"root_path = Path(\"/kaggle/input/rsna-intracranial-aneurysm-detection/series/\")\npatient_dirs = sorted([p for p in root_path.iterdir() if p.is_dir()])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:57:45.599865Z","iopub.execute_input":"2025-11-05T09:57:45.600249Z","iopub.status.idle":"2025-11-05T09:58:06.588092Z","shell.execute_reply.started":"2025-11-05T09:57:45.600223Z","shell.execute_reply":"2025-11-05T09:58:06.586983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_aneurysm_present_for_patient(patient_dir: Path, labels_df: pd.DataFrame):\n    \"\"\"\n    Given a patient directory Path (SeriesInstanceUID) and the labels DataFrame,\n    return the Aneurysm Present value (0 or 1).\n    \"\"\"\n    # Extract SeriesInstanceUID from folder name\n    series_uid = patient_dir.name\n    \n    # Filter DataFrame\n    result = labels_df.loc[labels_df[\"SeriesInstanceUID\"] == series_uid, \"Aneurysm Present\"]\n    \n    if not result.empty:\n        return int(result.values[0])\n    else:\n        # UID not found\n        return None\n        \n\ndef get_modality_for_patient(patient_dir: Path, labels_df: pd.DataFrame):\n    \"\"\"\n    Given a patient directory Path (SeriesInstanceUID) and the labels DataFrame,\n    return the Modality value (e.g., 'CT', 'MR', etc.).\n    \"\"\"\n    # Extract SeriesInstanceUID from folder name\n    series_uid = patient_dir.name\n    \n    # Filter DataFrame\n    result = labels_df.loc[labels_df[\"SeriesInstanceUID\"] == series_uid, \"Modality\"]\n    \n    if not result.empty:\n        return result.values[0]\n    else:\n        # UID not found\n        return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:58:06.589806Z","iopub.execute_input":"2025-11-05T09:58:06.590087Z","iopub.status.idle":"2025-11-05T09:58:06.598629Z","shell.execute_reply.started":"2025-11-05T09:58:06.590067Z","shell.execute_reply":"2025-11-05T09:58:06.597248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AneurysmSubjectDataset(Dataset):\n    \"\"\"\n    Loads full 3D volumes (stacked 2D slices) per patient, along with\n    the aneurysm label and modality information.\n\n    Args:\n        patient_dirs (list[Path]): list of SeriesInstanceUID directories\n        labels_df (pd.DataFrame): dataframe with labels\n        train (bool): whether to use train split or validation split\n        test_size (float): fraction of data to use as validation\n        transform: optional transform function (applied to 3D volume)\n        random_state (int): random seed for train/validation split\n    \"\"\"\n    def __init__(self, patient_dirs, labels_df, train=True, test_size=0.2, transform=None, random_state=42):\n        self.labels_df = labels_df\n        self.transform = transform\n\n        # Split into train and validation sets\n        train_dirs, val_dirs = train_test_split(\n            patient_dirs, test_size=test_size, random_state=random_state, shuffle=True\n        )\n        self.patient_dirs = train_dirs if train else val_dirs\n\n    def __len__(self):\n        return len(self.patient_dirs)\n\n    def __getitem__(self, idx):\n        patient_dir = self.patient_dirs[idx]\n\n        # --- Load DICOM slices ---\n        dicom_files = list(patient_dir.glob(\"*.dcm\"))\n        dicoms = [pydicom.dcmread(f) for f in dicom_files]\n        dicoms.sort(key=lambda dcm: int(dcm.InstanceNumber))\n\n        slices = [dcm.pixel_array for dcm in dicoms]\n\n        # Stack into 3D volume: [H, W, D]\n        volume = np.stack(slices, axis=-1)  # [H, W, D]\n        volume = torch.from_numpy(volume).unsqueeze(0).float()  # [1, H, W, D]\n    \n        if self.transform:\n            import torchio as tio\n            subject = tio.Subject(image=tio.ScalarImage(tensor=volume))\n            subject = self.transform(subject)\n            volume = subject.image.data  # Extract transformed tensor\n\n        # --- Labels and Metadata ---\n        label = get_aneurysm_present_for_patient(patient_dir, self.labels_df)\n        modality = get_modality_for_patient(patient_dir, self.labels_df)\n\n        label = torch.tensor(label, dtype=torch.long)\n        modality = str(modality)  # keep as string, not tensor\n\n        return {\n            \"image\": volume,\n            \"label\": label,\n            \"modality\": modality,\n        }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:58:17.770427Z","iopub.execute_input":"2025-11-05T09:58:17.770766Z","iopub.status.idle":"2025-11-05T09:58:17.784943Z","shell.execute_reply.started":"2025-11-05T09:58:17.770743Z","shell.execute_reply":"2025-11-05T09:58:17.783396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = AneurysmSubjectDataset(\n    patient_dirs=patient_dirs,\n    labels_df=labels,\n    train=True,\n    test_size=0.5  \n)\n\nval_dataset = AneurysmSubjectDataset(\n    patient_dirs=patient_dirs,\n    labels_df=labels,\n    train=False,\n    test_size=0.5\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:58:20.390976Z","iopub.execute_input":"2025-11-05T09:58:20.391547Z","iopub.status.idle":"2025-11-05T09:58:20.403565Z","shell.execute_reply.started":"2025-11-05T09:58:20.391518Z","shell.execute_reply":"2025-11-05T09:58:20.402173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample = train_dataset[1]\nimg = sample[\"image\"]\nlabel = sample[\"label\"]\nmodal = sample[\"modality\"]\n\nprint(\"Modality:\", modal)\nprint(\"Img Shape:\", img.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-05T09:59:16.638032Z","iopub.execute_input":"2025-11-05T09:59:16.639305Z","iopub.status.idle":"2025-11-05T09:59:20.167969Z","shell.execute_reply.started":"2025-11-05T09:59:16.639266Z","shell.execute_reply":"2025-11-05T09:59:20.166735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import defaultdict\nimport statistics as stats\nfrom tqdm import tqdm  # optional but nice for long loops\n\nshape_stats = defaultdict(list)\n\n# --- Collect shapes per modality ---\nfor i in tqdm(range(len(train_dataset)), desc=\"Analyzing dataset\"):\n    sample = train_dataset[i]\n    img = sample[\"image\"]\n    modality = sample[\"modality\"]\n\n    # store shape as tuple (C, H, W, D)\n    shape_stats[modality].append(tuple(img.shape))\n\n# --- Summarize ---\nfor modality, shapes in shape_stats.items():\n    print(f\"\\nModality: {modality}\")\n    print(f\"  Number of samples: {len(shapes)}\")\n\n    unique_shapes = set(shapes)\n    if len(unique_shapes) == 1:\n        print(f\"  All shapes: {list(unique_shapes)[0]}\")\n    else:\n        Hs = [s[1] for s in shapes]\n        Ws = [s[2] for s in shapes]\n        Ds = [s[3] for s in shapes]\n\n        print(f\"  Unique shapes: {len(unique_shapes)} distinct shapes\")\n\n        print(f\"  Height range: {min(Hs)} - {max(Hs)}\")\n        print(f\"  Width range:  {min(Ws)} - {max(Ws)}\")\n        print(f\"  Depth range:  {min(Ds)} - {max(Ds)}\")\n\n        print(f\"  Mean Height: {stats.mean(Hs):.1f}, Median: {stats.median(Hs)}\")\n        print(f\"  Mean Width:  {stats.mean(Ws):.1f}, Median: {stats.median(Ws)}\")\n        print(f\"  Mean Depth:  {stats.mean(Ds):.1f}, Median: {stats.median(Ds)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\n\nfor modality in shape_stats.keys():\n    # find first example of that modality\n    for i in range(len(train_dataset)):\n        s = train_dataset[i]\n        if s[\"modality\"] == modality:\n            img = s[\"image\"].squeeze().numpy()\n            mid_slice = img[..., img.shape[-1] // 2]\n            plt.imshow(mid_slice, cmap='gray')\n            plt.title(f\"{modality} - shape {tuple(s['image'].shape)}\")\n            plt.axis('off')\n            plt.show()\n            break","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}