{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13190393,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\nimport glob\nimport warnings\nimport pydicom\nimport seaborn as sns\nimport numpy as np\nimport pandas as pd\nimport nibabel as nib\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom random import sample\nfrom scipy.ndimage import zoom\n\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-05T02:23:30.358298Z","iopub.execute_input":"2025-08-05T02:23:30.359123Z","iopub.status.idle":"2025-08-05T02:23:35.429993Z","shell.execute_reply.started":"2025-08-05T02:23:30.359081Z","shell.execute_reply":"2025-08-05T02:23:35.429047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define paths\ntrain_csv_path = '/kaggle/input/rsna-intracranial-aneurysm-detection/train.csv'     \nlocalizers_csv_path = '/kaggle/input/rsna-intracranial-aneurysm-detection/train_localizers.csv'\nseries_dir = '/kaggle/input/rsna-intracranial-aneurysm-detection/series'                \n\n# Load train.csv as df\ndf = pd.read_csv(train_csv_path)\n# Load train_localizers.csv as loc_df\nloc_df = pd.read_csv(localizers_csv_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T02:23:35.431462Z","iopub.execute_input":"2025-08-05T02:23:35.432002Z","iopub.status.idle":"2025-08-05T02:23:35.497811Z","shell.execute_reply.started":"2025-08-05T02:23:35.43197Z","shell.execute_reply":"2025-08-05T02:23:35.496952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter for series where Aneurysm Present is 1\ndf_aneurysm = df[df['Aneurysm Present'] == 1]\n\n# Get unique SeriesInstanceUIDs from loc_df (series with localization data)\nloc_series = loc_df['SeriesInstanceUID'].unique()\n\n# Filter df_aneurysm to include only SeriesInstanceUIDs present in loc_df\ndf_aneurysm_loc = df_aneurysm[df_aneurysm['SeriesInstanceUID'].isin(loc_series)]\n\n# Extract relevant columns: SeriesInstanceUID and Modality\nmodality_map = df_aneurysm_loc[['SeriesInstanceUID', 'Modality']].drop_duplicates()\n\n# Initialize a dictionary to store modality to SeriesInstanceUID mappings\nmodality_series = {}\n\n# Iterate through each unique SeriesInstanceUID and its modality\nfor _, row in modality_map.iterrows():\n    series_uid = row['SeriesInstanceUID']\n    modality = row['Modality']\n    \n    # Store the SeriesInstanceUID for this modality\n    if modality not in modality_series:\n        modality_series[modality] = []\n    modality_series[modality].append(series_uid)\n\n# Prepare data for plotting\nmodality_counts = {modality: len(uids) for modality, uids in modality_series.items()}\nmodality_df = pd.DataFrame(list(modality_counts.items()), columns=['Modality', 'Count'])\n\n# Plot the number of series per modality\nplt.figure(figsize=(10, 6))\ng = sns.barplot(x='Modality', y='Count', data=modality_df, palette='viridis')\nplt.title('Number of Series with Aneurysm and Location per Modality')\nplt.xlabel('Modality')\nplt.ylabel('Number of Series')\nplt.xticks(rotation=45)\n\n# Add count labels on top of each bar\nfor p in g.patches:\n    g.annotate(f'{p.get_height():.0f}', (p.get_x() + p.get_width() / 2., p.get_height()), ha='center', va='center', xytext=(0, 9), textcoords='offset points')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T02:23:35.498619Z","iopub.execute_input":"2025-08-05T02:23:35.498934Z","iopub.status.idle":"2025-08-05T02:23:35.879926Z","shell.execute_reply.started":"2025-08-05T02:23:35.498906Z","shell.execute_reply":"2025-08-05T02:23:35.879055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"merged_df = loc_df.merge(df[['SeriesInstanceUID', 'Modality', 'Aneurysm Present']], on='SeriesInstanceUID', how='inner')\nmerged_df = merged_df[merged_df['Aneurysm Present'] == 1]\n\n# Group by Modality and location to count occurrences\nlabel_modality_counts = merged_df.groupby(['Modality', 'location']).size().reset_index(name='Count')\n\n# Plot as a grouped bar plot\nplt.figure(figsize=(12, 8))\ng = sns.barplot(x='location', y='Count', hue='Modality', data=label_modality_counts, palette='Set2')\nplt.title('Number of Aneurysms per Location and Modality')\nplt.xlabel('Aneurysm Location')\nplt.ylabel('Count')\nplt.xticks(rotation=90)\nplt.legend(title='Modality')\n\n# Add count labels on top of each bar\nfor p in g.patches:\n    height = p.get_height()\n    if height > 0:  # Only annotate non-zero bars\n        g.annotate(f'{height:.0f}', \n                   (p.get_x() + p.get_width() / 2., height), \n                   ha='center', va='center', \n                   xytext=(0, 5), textcoords='offset points', fontsize=8)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T02:23:35.88111Z","iopub.execute_input":"2025-08-05T02:23:35.881374Z","iopub.status.idle":"2025-08-05T02:23:36.67401Z","shell.execute_reply.started":"2025-08-05T02:23:35.881346Z","shell.execute_reply":"2025-08-05T02:23:36.67319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CMAP = 'gray'\n\ndef plot_labeled_slice(row_number, localizers_df, train_df, ax_image, ax_label, ax_overlay):\n    row = localizers_df.iloc[row_number]\n    series_uid = row['SeriesInstanceUID']\n    sop_uid = row['SOPInstanceUID']\n    coordinates = row['coordinates']\n    location = row['location']\n    \n    # Get Modality from train_df\n    modality = train_df[train_df['SeriesInstanceUID'] == series_uid]['Modality'].iloc[0] if not train_df[train_df['SeriesInstanceUID'] == series_uid].empty else 'Unknown'\n    \n    # Find the DICOM file for the SOPInstanceUID\n    folders = glob.glob(\"../input/rsna-intracranial-aneurysm-detection/series/*\")\n    series_path = None\n    for folder in folders:\n        if series_uid in folder:\n            series_path = folder\n            break\n    \n    if not series_path:\n        print(f\"No folder found for SeriesInstanceUID: {series_uid}\")\n        return\n    \n    # Get all DICOM files in the series\n    files = sorted(glob.glob(os.path.join(series_path, \"*.dcm\")))\n    dicom_file = None\n    for file in files:\n        if sop_uid in file:\n            dicom_file = file\n            break\n    \n    if not dicom_file:\n        print(f\"No DICOM file found for SOPInstanceUID: {sop_uid}\")\n        return\n    \n    # Load DICOM image\n    ex = pydicom.dcmread(dicom_file)\n    if len(ex.pixel_array.shape) > 2:\n        print(f\"Skipping {sop_uid}: Image has more than 2 dimensions\")\n        return\n    \n    img = ex.pixel_array\n    img = np.flipud(img)  # Flip vertically to match DICOM top-left origin\n    height, width = img.shape\n    \n    # Parse coordinates\n    coordinates = json.loads(coordinates.replace(\"'\", '\"'))\n    x = coordinates[\"x\"]\n    y = height - 1 - coordinates[\"y\"]  # Adjust for flipud\n    \n    # Plot Image\n    ax_image.imshow(img, cmap=CMAP, origin='upper', aspect=1)  # Use gray for DICOM\n    ax_image.set_title(f'Image\\nSeries: {series_uid[:8]}...')\n    ax_image.axis('off')\n    \n    # Plot Label (black background with red rectangle and text)\n    ax_label.imshow(np.zeros_like(img), cmap=CMAP, origin='upper', aspect=1)  # Black background\n    rect = patches.Rectangle((x-10, y-10), 20, 20, linewidth=1, edgecolor='r', facecolor='none')\n    ax_label.add_patch(rect)\n    ax_label.text(x, y+35, location, color='yellow', fontsize=8, ha='center', va='bottom')\n    ax_label.text(x, y+55, f'Modality: {modality}', color='red', fontsize=8, ha='center', va='bottom')\n    ax_label.set_title('Label')\n    ax_label.axis('off')\n    \n    # Plot Overlay\n    ax_overlay.imshow(img, cmap=CMAP, origin='upper', aspect=1)  # Use gray for DICOM\n    rect = patches.Rectangle((x-10, y-10), 20, 20, linewidth=1, edgecolor='r', facecolor='none')\n    ax_overlay.add_patch(rect)\n    ax_overlay.text(x, y+35, location, color='yellow', fontsize=8, ha='center', va='bottom')\n    ax_overlay.text(x, y+55, f'Modality: {modality}', color='red', fontsize=8, ha='center', va='bottom')\n    ax_overlay.set_title('Overlay')\n    ax_overlay.axis('off')\n\n# Function to plot for a given modality, optional label (location), and optional row indices\ndef plot_modality_slices(modality, label=None, row_indices=None):\n    # Filter for series with Aneurysm Present and localization data\n    df_aneurysm = df[df['Aneurysm Present'] == 1]\n    loc_series = loc_df['SeriesInstanceUID'].unique()\n    df_aneurysm_loc = df_aneurysm[df_aneurysm['SeriesInstanceUID'].isin(loc_series)]\n    df_modality = df_aneurysm_loc[df_aneurysm_loc['Modality'] == modality]\n    \n    # Get sorted unique SeriesInstanceUID for the modality\n    modality_series_sorted = sorted(df_modality['SeriesInstanceUID'].unique())\n    \n    # Select rows from loc_df for the modality\n    modality_loc_rows = loc_df[loc_df['SeriesInstanceUID'].isin(modality_series_sorted)]\n    \n    # Optionally filter by label (aneurysm location)\n    if label is not None:\n        modality_loc_rows = modality_loc_rows[modality_loc_rows['location'] == label]\n    \n    # If specific row indices are provided, try to use them\n    selected_rows = []\n    if row_indices is not None:\n        for idx in row_indices:\n            if idx < len(modality_loc_rows) and modality_loc_rows.iloc[idx]['SeriesInstanceUID'] in modality_series_sorted:\n                selected_rows.append(modality_loc_rows.index[idx])  # Use global index from loc_df\n    \n    # If fewer than 5 rows selected (or none), fall back to first 5 sorted series\n    if len(selected_rows) < 5:\n        if row_indices is not None:\n            print(f\"Requested indices {row_indices} invalid or insufficient. Selecting first {5 - len(selected_rows)} sorted series.\")\n        remaining_series = [s for s in modality_series_sorted if s not in [loc_df.iloc[idx]['SeriesInstanceUID'] for idx in selected_rows]]\n        for series in remaining_series[:5 - len(selected_rows)]:\n            series_rows = modality_loc_rows[modality_loc_rows['SeriesInstanceUID'] == series]\n            if not series_rows.empty:\n                selected_rows.append(series_rows.index[0])  # Pick first row for the series\n    \n    # If fewer than 5 series found, warn\n    if len(selected_rows) < 5:\n        print(f\"Only {len(selected_rows)} {modality} series with aneurysms and localization data found (label: {label}).\")\n    \n    # Create subplot grid (adjust rows based on available series)\n    fig, axs = plt.subplots(len(selected_rows), 3, figsize=(15, 5 * len(selected_rows)))\n    \n    # Handle case of single row (2D array to 1D)\n    if len(selected_rows) == 1:\n        axs = [axs]\n    \n    # Plot for each selected row\n    for i, row_number in enumerate(selected_rows):\n        print(f\"\\nDisplaying labeled slice for series {loc_df.iloc[row_number]['SeriesInstanceUID'][:8]}... (row {row_number})\")\n        plot_labeled_slice(row_number, loc_df, df, axs[i][0], axs[i][1], axs[i][2])\n\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T02:23:36.676626Z","iopub.execute_input":"2025-08-05T02:23:36.676992Z","iopub.status.idle":"2025-08-05T02:23:36.703207Z","shell.execute_reply.started":"2025-08-05T02:23:36.676963Z","shell.execute_reply":"2025-08-05T02:23:36.702168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplot_modality_slices(modality='CTA', label='Left Middle Cerebral Artery')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T02:23:36.70413Z","iopub.execute_input":"2025-08-05T02:23:36.704448Z","iopub.status.idle":"2025-08-05T02:23:37.422409Z","shell.execute_reply.started":"2025-08-05T02:23:36.704407Z","shell.execute_reply":"2025-08-05T02:23:37.42147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_modality_slices(modality='CTA')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T02:23:37.423381Z","iopub.execute_input":"2025-08-05T02:23:37.423708Z","iopub.status.idle":"2025-08-05T02:23:39.928568Z","shell.execute_reply.started":"2025-08-05T02:23:37.42366Z","shell.execute_reply":"2025-08-05T02:23:39.927664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_modality_slices(modality='MRA')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T02:23:39.929511Z","iopub.execute_input":"2025-08-05T02:23:39.929804Z","iopub.status.idle":"2025-08-05T02:23:42.756077Z","shell.execute_reply.started":"2025-08-05T02:23:39.929782Z","shell.execute_reply":"2025-08-05T02:23:42.75485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_modality_slices(modality='MRI T2', row_indices=[21, 40, 48, 49])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T02:23:42.75716Z","iopub.execute_input":"2025-08-05T02:23:42.757427Z","iopub.status.idle":"2025-08-05T02:23:44.935969Z","shell.execute_reply.started":"2025-08-05T02:23:42.757397Z","shell.execute_reply":"2025-08-05T02:23:44.934868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_modality_slices(modality='MRI T1post')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T02:23:44.936935Z","iopub.execute_input":"2025-08-05T02:23:44.937175Z","iopub.status.idle":"2025-08-05T02:23:47.490216Z","shell.execute_reply.started":"2025-08-05T02:23:44.937158Z","shell.execute_reply":"2025-08-05T02:23:47.488936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.Modality.unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-05T02:23:47.491356Z","iopub.execute_input":"2025-08-05T02:23:47.491678Z","iopub.status.idle":"2025-08-05T02:23:47.499531Z","shell.execute_reply.started":"2025-08-05T02:23:47.491654Z","shell.execute_reply":"2025-08-05T02:23:47.498749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}