{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":36363,"databundleVersionId":4050810,"sourceType":"competition"}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nRSNA Dataset Extractor for Gaussian Splatting Pipeline\nSimple script to extract sample DICOM files from RSNA 2022 Cervical Spine dataset\n\"\"\"\n\nimport os\nimport shutil\nimport json\nimport pydicom\nfrom pathlib import Path\nfrom tqdm import tqdm\n\ndef extract_rsna_samples(dataset_path, output_path, num_samples=5):\n    \"\"\"\n    Extract sample DICOM studies from RSNA dataset for Gaussian Splatting pipeline\n    \n    Args:\n        dataset_path (str): Path to RSNA dataset directory\n        output_path (str): Output directory for extracted samples\n        num_samples (int): Number of sample studies to extract\n    \"\"\"\n    \n    # Create output directory\n    os.makedirs(output_path, exist_ok=True)\n    \n    # Find all study directories\n    train_path = os.path.join(dataset_path, \"train_images\")\n    if not os.path.exists(train_path):\n        print(f\"Error: Dataset path not found: {train_path}\")\n        return\n    \n    study_dirs = [d for d in os.listdir(train_path) if os.path.isdir(os.path.join(train_path, d))]\n    print(f\"Found {len(study_dirs)} studies in dataset\")\n    \n    # Load metadata for sample selection\n    metadata_path = os.path.join(dataset_path, \"train.csv\")\n    metadata = {}\n    if os.path.exists(metadata_path):\n        import pandas as pd\n        df = pd.read_csv(metadata_path)\n        # Group by StudyInstanceUID for fracture information\n        study_fractures = df.groupby('StudyInstanceUID')['patient_overall'].first().to_dict()\n        metadata = study_fractures\n        print(f\"Loaded metadata for {len(metadata)} studies\")\n    \n    # Select diverse samples (mix of healthy and fractured cases)\n    selected_studies = []\n    healthy_count = 0\n    fractured_count = 0\n    \n    for study_id in study_dirs[:50]:  # Check first 50 studies\n        fracture_status = metadata.get(study_id, 0)\n        \n        if fracture_status == 0 and healthy_count < num_samples // 2:\n            selected_studies.append((study_id, \"healthy\"))\n            healthy_count += 1\n        elif fracture_status == 1 and fractured_count < num_samples // 2:\n            selected_studies.append((study_id, \"fractured\"))\n            fractured_count += 1\n        \n        if len(selected_studies) >= num_samples:\n            break\n    \n    # Fill remaining slots if needed\n    remaining = num_samples - len(selected_studies)\n    for study_id in study_dirs[50:]:\n        if remaining <= 0:\n            break\n        selected_studies.append((study_id, \"unknown\"))\n        remaining -= 1\n    \n    print(f\"Selected {len(selected_studies)} studies for extraction\")\n    \n    # Extract selected studies\n    for i, (study_id, status) in enumerate(tqdm(selected_studies, desc=\"Extracting studies\")):\n        study_path = os.path.join(train_path, study_id)\n        output_study_path = os.path.join(output_path, f\"study_{i+1:02d}_{status}\")\n        \n        # Copy entire study directory\n        if os.path.exists(study_path):\n            shutil.copytree(study_path, output_study_path, dirs_exist_ok=True)\n            \n            # Validate DICOM files and add .dcm extension if missing\n            validate_and_fix_dicom_files(output_study_path)\n            \n            # Create study info file\n            study_info = {\n                \"study_id\": study_id,\n                \"status\": status,\n                \"original_path\": study_path,\n                \"extracted_path\": output_study_path,\n                \"dicom_count\": count_dicom_files(output_study_path)\n            }\n            \n            with open(os.path.join(output_study_path, \"study_info.json\"), \"w\") as f:\n                json.dump(study_info, f, indent=2)\n            \n            print(f\"Extracted study {study_id} ({status}) -> {study_info['dicom_count']} DICOM files\")\n\ndef validate_and_fix_dicom_files(study_path):\n    \"\"\"\n    Validate DICOM files and add .dcm extension if missing\n    \"\"\"\n    for root, dirs, files in os.walk(study_path):\n        for file in files:\n            file_path = os.path.join(root, file)\n            \n            # Skip if already has .dcm extension\n            if file.endswith('.dcm'):\n                continue\n            \n            # Try to read as DICOM\n            try:\n                ds = pydicom.dcmread(file_path, force=True)\n                # If successful, add .dcm extension\n                new_path = file_path + '.dcm'\n                os.rename(file_path, new_path)\n            except:\n                # Not a valid DICOM file, skip\n                continue\n\ndef count_dicom_files(study_path):\n    \"\"\"Count DICOM files in study directory\"\"\"\n    count = 0\n    for root, dirs, files in os.walk(study_path):\n        for file in files:\n            if file.endswith('.dcm') or is_dicom_file(os.path.join(root, file)):\n                count += 1\n    return count\n\ndef is_dicom_file(file_path):\n    \"\"\"Check if file is a valid DICOM file\"\"\"\n    try:\n        pydicom.dcmread(file_path, force=True)\n        return True\n    except:\n        return False\n\ndef main():\n    \"\"\"Main function to run the extractor\"\"\"\n    \n    # Configuration\n    DATASET_PATH = \"/kaggle/input/rsna-2022-cervical-spine-fracture-detection\"\n    OUTPUT_PATH = \"/kaggle/output/\"\n    \n    if not OUTPUT_PATH:\n        OUTPUT_PATH = \"./extracted_samples\"\n    \n    try:\n        NUM_SAMPLES = 5\n    except ValueError:\n        NUM_SAMPLES = 5\n    \n    print(f\"\\nExtracting {NUM_SAMPLES} samples from {DATASET_PATH} to {OUTPUT_PATH}\")\n    print(\"This will copy DICOM files and prepare them for the Gaussian Splatting pipeline\\n\")\n    \n    # Run extraction\n    extract_rsna_samples(DATASET_PATH, OUTPUT_PATH, NUM_SAMPLES)\n    \n    print(f\"\\n✅ Extraction complete! Sample studies saved to: {OUTPUT_PATH}\")\n    print(\"\\nNext steps for Gaussian Splatting pipeline:\")\n    print(\"1. Load DICOM files using pydicom\")\n    print(\"2. Convert to NIfTI format using nibabel\")\n    print(\"3. Run TotalSegmentator for vertebrae segmentation\")\n    print(\"4. Generate mesh using marching cubes\")\n    print(\"5. Create point cloud from mesh\")\n    print(\"6. Initialize and train Gaussian Splatting model\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"_uuid":"27e5434b-9f52-4511-847a-b75606ea505b","_cell_guid":"a0a31063-1207-48a2-98ac-54d2f0c97549","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2026-01-22T18:33:22.478813Z","iopub.execute_input":"2026-01-22T18:33:22.47918Z","iopub.status.idle":"2026-01-22T18:33:54.752177Z","shell.execute_reply.started":"2026-01-22T18:33:22.479149Z","shell.execute_reply":"2026-01-22T18:33:54.751165Z"}},"outputs":[],"execution_count":null}]}