{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":99552,"databundleVersionId":13851420,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":266316118,"sourceType":"kernelVersion"},{"sourceId":139093,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":117776,"modelId":141013}],"dockerImageVersionId":31154,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pydicom\nimport os\nimport glob\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Set the base path to your data\nBASE_PATH = '/kaggle/input/rsna-intracranial-aneurysm-detection/'\n\nprint(\"Setup complete. Libraries imported and base path set.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-07T10:38:32.582159Z","iopub.execute_input":"2025-10-07T10:38:32.582633Z","iopub.status.idle":"2025-10-07T10:38:34.101861Z","shell.execute_reply.started":"2025-10-07T10:38:32.582609Z","shell.execute_reply":"2025-10-07T10:38:34.101109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the training labels and localizer data\ndf_train = pd.read_csv(os.path.join(BASE_PATH, 'train.csv'))\ndf_localizers = pd.read_csv(os.path.join(BASE_PATH, 'train_localizers.csv'))\n\nprint(\"train.csv shape:\", df_train.shape)\nprint(\"train_localizers.csv shape:\", df_localizers.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T10:38:34.103034Z","iopub.execute_input":"2025-10-07T10:38:34.103365Z","iopub.status.idle":"2025-10-07T10:38:34.157158Z","shell.execute_reply.started":"2025-10-07T10:38:34.103347Z","shell.execute_reply":"2025-10-07T10:38:34.156586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display the first 5 rows of the training data\nprint(\"First 5 rows of train.csv:\")\ndisplay(df_train.head())\n\n# Analyze the distribution of the main target variable 'Aneurysm Present'\nprint(\"\\nDistribution of 'Aneurysm Present':\")\naneurysm_counts = df_train['Aneurysm Present'].value_counts()\nprint(aneurysm_counts)\n\n# Plot the distribution\nplt.figure(figsize=(6, 4))\nsns.countplot(x='Aneurysm Present', data=df_train)\nplt.title('Distribution of Aneurysm Presence')\nplt.xlabel('Aneurysm Present (1 = Yes, 0 = No)')\nplt.ylabel('Number of Scan Series')\nplt.xticks(ticks=[0, 1], labels=['No', 'Yes'])\nplt.show()\n\n# Print the percentage\nprint(f\"\\nPercentage of series with an aneurysm: {aneurysm_counts[1] / len(df_train) * 100:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T10:38:34.157821Z","iopub.execute_input":"2025-10-07T10:38:34.158072Z","iopub.status.idle":"2025-10-07T10:38:34.40228Z","shell.execute_reply.started":"2025-10-07T10:38:34.158055Z","shell.execute_reply":"2025-10-07T10:38:34.401529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display the first 5 rows of the localizers data\nprint(\"First 5 rows of train_localizers.csv:\")\ndisplay(df_localizers.head())\n\n# Merge the localizer data with the main training dataframe\n# This will give us a dataframe containing only the positive cases with their coordinates\ndf_merged = pd.merge(df_train, df_localizers, on='SeriesInstanceUID', how='inner')\n\nprint(f\"\\nShape of merged dataframe: {df_merged.shape}\")\nprint(\"This shape should match the number of rows in train_localizers.csv, which is 2254.\")\n\nprint(\"\\nFirst 5 rows of the merged dataframe (train + localizers):\")\ndisplay(df_merged.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T10:38:34.403956Z","iopub.execute_input":"2025-10-07T10:38:34.404656Z","iopub.status.idle":"2025-10-07T10:38:34.433936Z","shell.execute_reply.started":"2025-10-07T10:38:34.404627Z","shell.execute_reply":"2025-10-07T10:38:34.43319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import ast\n\n# The 'coordinates' column is a string representation of a dictionary.\n# We use ast.literal_eval to safely convert it into an actual dictionary.\ndf_merged['coordinates_dict'] = df_merged['coordinates'].apply(ast.literal_eval)\n\n# Now, create separate columns for 'x' and 'y' coordinates.\ndf_merged['x'] = df_merged['coordinates_dict'].apply(lambda d: d['x'])\ndf_merged['y'] = df_merged['coordinates_dict'].apply(lambda d: d['y'])\n\n# We can drop the intermediate columns now\ndf_merged = df_merged.drop(columns=['coordinates', 'coordinates_dict'])\n\nprint(\"Parsed 'x' and 'y' coordinates.\")\ndisplay(df_merged[['SeriesInstanceUID', 'SOPInstanceUID', 'x', 'y', 'location']].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T10:38:34.434876Z","iopub.execute_input":"2025-10-07T10:38:34.43512Z","iopub.status.idle":"2025-10-07T10:38:34.482481Z","shell.execute_reply.started":"2025-10-07T10:38:34.435091Z","shell.execute_reply":"2025-10-07T10:38:34.481884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select the first sample from our merged dataframe\nsample = df_merged.iloc[5]\n\nseries_uid = sample['SeriesInstanceUID']\nsop_uid = sample['SOPInstanceUID']\nx_coord = sample['x']\ny_coord = sample['y']\nlocation_label = sample['location']\n\n# Construct the full path to the specific DICOM file\ndcm_path = os.path.join(BASE_PATH, 'series', series_uid, f\"{sop_uid}.dcm\")\nprint(f\"Loading DICOM file from: {dcm_path}\")\n\n# Load the DICOM file using pydicom\ndcm_file = pydicom.dcmread(dcm_path)\n\n# Get the pixel data from the DICOM file\nimage = dcm_file.pixel_array\n\n# Display the image\nplt.figure(figsize=(10, 10))\nplt.imshow(image, cmap='gray')\n\n# Plot a red circle on top of the image at the aneurysm's coordinates\nplt.scatter(x_coord, y_coord, s=200, facecolors='none', edgecolors='r', linewidth=2)\n\nplt.title(f\"Aneurysm Location: '{location_label}'\\n at ({x_coord:.1f}, {y_coord:.1f})\", fontsize=14)\nplt.axis('off') # Hide the x and y axes for a cleaner look\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T10:58:46.193017Z","iopub.execute_input":"2025-10-07T10:58:46.193541Z","iopub.status.idle":"2025-10-07T10:58:46.505261Z","shell.execute_reply.started":"2025-10-07T10:58:46.193516Z","shell.execute_reply":"2025-10-07T10:58:46.50438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom tqdm.notebook import tqdm\nimport cv2\n\n# --- Configuration ---\nSAMPLE_FRACTION = 0.1 \nBBOX_SIZE = 30\nOUTPUT_DIR = '/kaggle/working/dataset'\n\n# --- Create Directory Structure ---\nfor split in ['train', 'val']:\n    os.makedirs(os.path.join(OUTPUT_DIR, 'images', split), exist_ok=True)\n    os.makedirs(os.path.join(OUTPUT_DIR, 'labels', split), exist_ok=True)\n\n# --- Sample and Split Data ---\ndata_sample = df_merged.sample(frac=SAMPLE_FRACTION, random_state=42)\ntrain_df, val_df = train_test_split(data_sample, test_size=0.2, random_state=42, stratify=data_sample['location'])\n\nprint(f\"Processing {len(train_df)} images for training and {len(val_df)} for validation.\")\n\n# --- Processing Function ---\ndef process_and_save_sample(df, split):\n    \"\"\"Reads DICOM, robustly converts to 2D grayscale PNG, and creates YOLO label file.\"\"\"\n    success_count = 0\n    error_count = 0\n    for _, row in tqdm(df.iterrows(), total=len(df), desc=f'Processing {split} set'):\n        dcm_path = os.path.join(BASE_PATH, 'series', row['SeriesInstanceUID'], f\"{row['SOPInstanceUID']}.dcm\")\n        image_filename = f\"{row['SOPInstanceUID']}.png\"\n        label_filename = f\"{row['SOPInstanceUID']}.txt\"\n        image_save_path = os.path.join(OUTPUT_DIR, 'images', split, image_filename)\n        label_save_path = os.path.join(OUTPUT_DIR, 'labels', split, label_filename)\n\n        try:\n            # 1. Read DICOM and get pixel array\n            dcm = pydicom.dcmread(dcm_path)\n            image = dcm.pixel_array\n\n            # --- ROBUSTNESS FIX STARTS HERE ---\n            # Handle different image dimensions and channels\n            if image.ndim == 3:\n                # Case 1: (frames, height, width) -> multi-frame grayscale\n                # Heuristic: take the middle frame\n                if image.shape[0] > 1 and image.shape[2] > 4: # check if last dim is not channels\n                     middle_slice_idx = image.shape[0] // 2\n                     image = image[middle_slice_idx]\n                # Case 2: (height, width, channels) -> color image\n                elif image.shape[2] in [3, 4]:\n                    image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n                # Case 3: (1, height, width) -> single-frame in a 3D array\n                elif image.shape[0] == 1:\n                    image = image[0]\n            \n            if image.ndim != 2:\n                 raise ValueError(f\"Image array not successfully converted to 2D. Shape is {image.shape}\")\n            # --- ROBUSTNESS FIX ENDS HERE ---\n\n            # 2. Normalize to 0-255 and convert to 8-bit unsigned integer\n            image = cv2.normalize(image, None, 0, 255, cv2.NORM_MINMAX).astype(np.uint8)\n\n            # 3. Save as PNG\n            cv2.imwrite(image_save_path, image)\n\n            # 4. Create YOLO label file\n            img_h, img_w = image.shape\n            x_center_norm = row['x'] / img_w\n            y_center_norm = row['y'] / img_h\n            width_norm = BBOX_SIZE / img_w\n            height_norm = BBOX_SIZE / img_h\n            \n            with open(label_save_path, 'w') as f:\n                f.write(f\"0 {x_center_norm} {y_center_norm} {width_norm} {height_norm}\\n\")\n            \n            success_count += 1\n        \n        except Exception as e:\n            # print(f\"Skipping file {dcm_path} due to error: {e}\")\n            error_count += 1\n    \n    print(f\"\\nFinished processing {split} set.\")\n    print(f\"Successfully processed: {success_count}\")\n    print(f\"Failed to process: {error_count}\")\n\n# --- Run Processing ---\nprocess_and_save_sample(train_df, 'train')\nprocess_and_save_sample(val_df, 'val')\n\nprint(\"\\nData preparation complete. Check the '/kaggle/working/dataset' directory.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T10:38:34.809121Z","iopub.execute_input":"2025-10-07T10:38:34.809608Z","iopub.status.idle":"2025-10-07T10:38:55.136535Z","shell.execute_reply.started":"2025-10-07T10:38:34.809586Z","shell.execute_reply":"2025-10-07T10:38:55.135768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import yaml\nimport os\nimport sys\nimport glob\n\n# --- Part 1: Create the dataset.yaml file (Unchanged) ---\ndataset_config = {\n    'path': '/kaggle/working/dataset',\n    'train': 'images/train',\n    'val': 'images/val',\n    'names': { 0: 'aneurysm' }\n}\nwith open('dataset.yaml', 'w') as f:\n    yaml.dump(dataset_config, f, default_flow_style=False)\nprint(\"--- dataset.yaml created ---\")\nwith open('dataset.yaml', 'r') as f: print(f.read())\nprint(\"--------------------------\")\n\n# --- Part 2: Bypassing pip by extracting the .whl file directly from your dataset ---\nprint(\"\\nBypassing pip. Using ultralytics by extracting the wheel file.\")\n\n# 1. Define the path to your dataset's 'packages' directory.\npackages_dir = '/kaggle/input/queryplanner-ultralytics-for-offline-install/packages'\n\n# 2. Find the exact path to the ultralytics wheel file inside that directory.\ntry:\n    ultralytics_whl_path = glob.glob(f'{packages_dir}/ultralytics-*.whl')[0]\n    print(f\"Found wheel file: {ultralytics_whl_path}\")\nexcept IndexError:\n    raise FileNotFoundError(\"Could not find the ultralytics wheel file in your dataset's 'packages' directory.\")\n\n# 3. A .whl file is just a zip file. Unzip it to extract the source code.\n#    We will extract it into the current working directory.\n!unzip -q {ultralytics_whl_path} -d /kaggle/working/\n\n# 4. The source code is now in a folder named 'ultralytics'. Add this to the Python path.\n#    This makes the module importable without any installation.\nsource_code_path = '/kaggle/working/ultralytics'\nsys.path.insert(0, source_code_path)\nprint(f\"Added {source_code_path} to Python path.\")\n\n# Now, the import will work.\nfrom ultralytics import YOLO\n\n# --- Part 3: Load the correct YOLOv8 model ---\nmodel_weights_path = '/kaggle/input/yolov8/pytorch/default/1/yolov8s.pt'\nprint(f\"Loading pre-trained YOLOv8 weights from: {model_weights_path}\")\nmodel = YOLO(model_weights_path)\n\n# --- Part 4: Train the Model ---\nprint(\"\\nStarting model training...\")\nresults = model.train(\n    data='dataset.yaml',\n    epochs=25,\n    imgsz=512,\n    batch=8,\n    project='runs',\n    name='aneurysm_detection_v8_final'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T10:47:18.663309Z","iopub.execute_input":"2025-10-07T10:47:18.663697Z","iopub.status.idle":"2025-10-07T10:52:12.735556Z","shell.execute_reply.started":"2025-10-07T10:47:18.663673Z","shell.execute_reply":"2025-10-07T10:52:12.734689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nimport polars as pl\nfrom ultralytics import YOLO\n\n# --- Load our trained model ---\n# The path points to the best weights saved during training\nMODEL_PATH = '/kaggle/working/runs/aneurysm_detection_v8_final/weights/best.pt'\nmodel = YOLO(MODEL_PATH)\n\n# --- Define Constants ---\nID_COL = 'SeriesInstanceUID'\nLABEL_COLS = [\n    'Left Infraclinoid Internal Carotid Artery', 'Right Infraclinoid Internal Carotid Artery',\n    'Left Supraclinoid Internal Carotid Artery', 'Right Supraclinoid Internal Carotid Artery',\n    'Left Middle Cerebral Artery', 'Right Middle Cerebral Artery', 'Anterior Communicating Artery',\n    'Left Anterior Cerebral Artery', 'Right Anterior Cerebral Artery', 'Left Posterior Communicating Artery',\n    'Right Posterior Communicating Artery', 'Basilar Tip', 'Other Posterior Circulation', 'Aneurysm Present',\n]\nCONFIDENCE_THRESHOLD = 0.05 # A low threshold to be safe\n\n# --- The Main Prediction Function ---\ndef predict(series_path: str) -> pl.DataFrame:\n    \"\"\"\n    Takes a path to a DICOM series, runs inference on each slice,\n    and returns a prediction for the entire series.\n    \"\"\"\n    series_id = os.path.basename(series_path)\n    aneurysm_detected_in_series = False\n\n    # Get all DICOM file paths in the series\n    dcm_files = glob.glob(os.path.join(series_path, '*.dcm'))\n\n    # Loop through each slice and run prediction\n    for dcm_path in dcm_files:\n        # We don't need the try-except block here because the test set should be clean,\n        # but it's good practice.\n        try:\n            dcm = pydicom.dcmread(dcm_path)\n            image = dcm.pixel_array\n\n            # Robustly convert to 2D grayscale, same logic as data prep\n            if image.ndim == 3:\n                if image.shape[0] > 1 and image.shape[2] > 4:\n                     image = image[image.shape[0] // 2]\n                elif image.shape[2] in [3, 4]:\n                    image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n                elif image.shape[0] == 1:\n                    image = image[0]\n            \n            # Run YOLO model inference\n            results = model.predict(image, imgsz=512, verbose=False)\n            \n            # Check if any detection has confidence above our threshold\n            if len(results[0].boxes) > 0:\n                if results[0].boxes.conf[0].item() > CONFIDENCE_THRESHOLD:\n                    aneurysm_detected_in_series = True\n                    break # Stop processing this series as we've found one\n        except Exception as e:\n            # Silently ignore problematic slices in test data\n            pass\n            \n    # --- Create the submission DataFrame ---\n    # For this baseline, if we detect *any* aneurysm, we'll give a high probability to 'Aneurysm Present'\n    # and a low, non-zero probability to all specific locations.\n    if aneurysm_detected_in_series:\n        aneurysm_present_prob = 0.9\n        location_prob = 0.1 # A small guess for all locations\n    else:\n        aneurysm_present_prob = 0.1\n        location_prob = 0.01\n\n    # Create a dictionary for the row data\n    data = {ID_COL: [series_id]}\n    for col in LABEL_COLS:\n        if col == 'Aneurysm Present':\n            data[col] = [aneurysm_present_prob]\n        else:\n            data[col] = [location_prob]\n\n    # Create a Polars DataFrame\n    predictions = pl.DataFrame(data)\n\n    # --- Important Cleanup Step for Kaggle ---\n    shutil.rmtree('/kaggle/shared', ignore_errors=True)\n\n    return predictions.drop(ID_COL)\n\nprint(\"Submission logic is ready.\")\n\n# This part of the code is for generating a submission.csv when you run the notebook directly.\n# The actual competition environment will use the `predict` function differently.\n# We'll simulate it with one of our validation images' series.\nif not os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    print(\"\\nRunning a local test...\")\n    \n    # Get a sample series path from our validation set\n    sample_series_path = os.path.join(BASE_PATH, 'series', val_df.iloc[0]['SeriesInstanceUID'])\n    \n    # Create a dummy submission file by calling our function\n    sample_prediction = predict(sample_series_path)\n    \n    # Add the ID column back for the local file\n    final_df = pl.DataFrame({ID_COL: [os.path.basename(sample_series_path)]}).hstack(sample_prediction)\n    \n    final_df.write_parquet('submission.parquet')\n    print(\"Created a sample submission.parquet file:\")\n    display(final_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T11:03:13.599946Z","iopub.execute_input":"2025-10-07T11:03:13.600845Z","iopub.status.idle":"2025-10-07T11:03:16.22625Z","shell.execute_reply.started":"2025-10-07T11:03:13.600819Z","shell.execute_reply":"2025-10-07T11:03:16.225563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}