{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99552,"databundleVersionId":13190393,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ==========================================\n# RSNA Intracranial Aneurysm Detection - Baseline Inference Code\n# ==========================================\n\n# Install necessary libraries\n!pip install polars -q      # High-performance DataFrame library\n!pip install pydicom -q     # For handling DICOM medical images\n\n# Import required libraries\nimport kaggle_evaluation.rsna_inference_server   # Kaggle inference server for competition submissions\nfrom collections import defaultdict              # Useful for dictionary-like structures\nimport pydicom                                   # For DICOM medical image processing\nimport shutil                                    # For file and directory operations\nimport os                                        # For path operations\nimport pandas as pd                              # For handling tabular data\nimport polars as pl                              # Alternative fast DataFrame library\nimport numpy as np                               # Numerical operations\nfrom sklearn import *                            # For machine learning utilities (if needed)\n\n\n# ==========================================\n# Step 1: Define paths and load dataset\n# ==========================================\n\n# Path to the dataset\ndata_path = '/kaggle/input/rsna-intracranial-aneurysm-detection/'\n\n# Read training metadata\ntrain_data = pd.read_csv(os.path.join(data_path, 'train.csv'))\ntrain_localizers = pd.read_csv(os.path.join(data_path, 'train_localizers.csv'))\n\n\n# ==========================================\n# Step 2: Define column names\n# ==========================================\n\n# Unique identifier for each series\nID_COL = 'SeriesInstanceUID'\n\n# Labels for different aneurysm locations\nLABEL_COLS = [\n    'Left Infraclinoid Internal Carotid Artery',\n    'Right Infraclinoid Internal Carotid Artery',\n    'Left Supraclinoid Internal Carotid Artery',\n    'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery',\n    'Right Middle Cerebral Artery',\n    'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery',\n    'Right Anterior Cerebral Artery',\n    'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery',\n    'Basilar Tip',\n    'Other Posterior Circulation',\n    'Aneurysm Present',\n]\n\n\n# ==========================================\n# Step 3: Define allowed DICOM tags\n# ==========================================\n\n# These metadata fields from DICOM files are considered safe to use\nDICOM_TAG_ALLOWLIST = [\n    'BitsAllocated', 'BitsStored', 'Columns', 'FrameOfReferenceUID', 'HighBit',\n    'ImageOrientationPatient', 'ImagePositionPatient', 'InstanceNumber', 'Modality',\n    'PatientID', 'PhotometricInterpretation', 'PixelRepresentation', 'PixelSpacing',\n    'PlanarConfiguration', 'RescaleIntercept', 'RescaleSlope', 'RescaleType',\n    'Rows', 'SOPClassUID', 'SOPInstanceUID', 'SamplesPerPixel', 'SliceThickness',\n    'SpacingBetweenSlices', 'StudyInstanceUID', 'TransferSyntaxUID',\n]\n\n\n# ==========================================\n# Step 4: Simple Baseline Model - Mean Strategy\n# ==========================================\n\n# Compute mean values of labels across the training dataset\n# This will be used as a naive prediction (always predicting class mean probability)\nmeans = train_data[LABEL_COLS].mean().to_dict()\n\n\n# ==========================================\n# Step 5: Define Prediction Function\n# ==========================================\n\ndef predict(series_path: str):\n    \"\"\"\n    Prediction function for RSNA Aneurysm Detection.\n\n    Args:\n        series_path (str): Path to the DICOM series folder.\n\n    Returns:\n        pl.DataFrame: Predictions for the given series without the ID column.\n    \"\"\"\n    \n    # Extract series ID from folder name\n    series_id = os.path.basename(series_path)\n\n    # Create a prediction DataFrame with mean values\n    predictions = pl.DataFrame(\n        data=[[series_id] + [means[k] for k in LABEL_COLS]],\n        schema=[ID_COL, *LABEL_COLS],\n        orient='row'\n    )\n\n    # Validate that predictions have the correct format\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == [ID_COL, *LABEL_COLS], \"Invalid Polars DataFrame schema\"\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == [ID_COL, *LABEL_COLS]).all(), \"Invalid Pandas DataFrame schema\"\n    else:\n        raise TypeError('The predict function must return a DataFrame (Polars or Pandas).')\n\n    # Clean up temporary shared directory (used by Kaggle inference server)\n    shutil.rmtree('/kaggle/shared', ignore_errors=True)\n\n    # Return predictions (excluding ID column, as required)\n    return predictions.drop(ID_COL)\n\n\n# ==========================================\n# Step 6: Initialize Inference Server\n# ==========================================\n\n# Initialize the RSNA inference server with our predict() function\ninference_server = kaggle_evaluation.rsna_inference_server.RSNAInferenceServer(predict)\n\n\n# ==========================================\n# Step 7: Run Inference\n# ==========================================\n\n# If running in competition mode, start the server\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\n\n# Otherwise, run locally for debugging\nelse:\n    inference_server.run_local_gateway()\n\n    # Display generated submission file for verification\n    display(pl.read_parquet('/kaggle/working/submission.parquet'))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-08-19T08:41:03.363556Z","iopub.execute_input":"2025-08-19T08:41:03.363988Z","iopub.status.idle":"2025-08-19T08:41:20.730833Z","shell.execute_reply.started":"2025-08-19T08:41:03.363963Z","shell.execute_reply":"2025-08-19T08:41:20.730098Z"}},"outputs":[],"execution_count":null}]}