{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":36363,"databundleVersionId":4050810}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install python-gdcm pylibjpeg pylibjpeg-libjpeg -q\nimport os\nimport numpy as np\nimport nibabel as nib\nimport pydicom\nfrom pathlib import Path\n\n# Output directory\noutput_dir = Path('/kaggle/working/cervical_nnunet')\nimages_dir = output_dir / 'imagesTr'\nlabels_dir = output_dir / 'labelsTr'\nimages_dir.mkdir(parents=True, exist_ok=True)\nlabels_dir.mkdir(parents=True, exist_ok=True)\n\nseg_dir = Path('/kaggle/input/competitions/rsna-2022-cervical-spine-fracture-detection/segmentations')\ntrain_images_dir = Path('/kaggle/input/competitions/rsna-2022-cervical-spine-fracture-detection/train_images')\n\n# Get all segmented study UIDs\nseg_files = list(seg_dir.glob('*.nii'))\nprint(f\"Found {len(seg_files)} segmentation files\")\n\ncase_idx = 0\nskipped = 0\n\nfor seg_path in seg_files:\n    uid = seg_path.stem  # e.g. 1.2.826.0.1.3680043.10633\n    study_dir = train_images_dir / uid\n    \n    if not study_dir.exists():\n        print(f\"Skipping {uid} — no training images found\")\n        skipped += 1\n        continue\n    \n    try:\n        # Load segmentation\n        seg_nib = nib.load(str(seg_path))\n        seg_array = seg_nib.get_fdata().astype(np.uint8)\n        \n        # Check it has cervical labels (1-7)\n        unique = np.unique(seg_array)\n        has_cervical = any(l in unique for l in range(1, 8))\n        if not has_cervical:\n            print(f\"Skipping {uid} — no cervical labels\")\n            skipped += 1\n            continue\n        \n        # Remap: keep 1-7 as is (C1-C7), remap 8-11 (T1-T4) to 0\n        # We only want cervical for our model\n        cervical_mask = np.zeros_like(seg_array)\n        for label in range(1, 12):  # 1-11 inclusive\n            cervical_mask[seg_array == label] = label\n        \n        # Load DICOM slices and convert to NIfTI\n        dicom_files = sorted(study_dir.glob('**/*.dcm'))\n        if len(dicom_files) == 0:\n            # Some studies have slices in subdirectories\n            dicom_files = sorted(study_dir.rglob('*.dcm'))\n        \n        if len(dicom_files) == 0:\n            print(f\"Skipping {uid} — no DICOM files\")\n            skipped += 1\n            continue\n        \n        print(f\"Processing {uid}: {len(dicom_files)} slices\")\n        \n        # Read all slices\n        slices = []\n        positions = []\n        for dcm_path in dicom_files:\n            ds = pydicom.dcmread(str(dcm_path))\n            slices.append(ds)\n            positions.append(float(ds.ImagePositionPatient[2]))\n        \n        # Sort by position\n        sorted_pairs = sorted(zip(positions, slices), key=lambda x: x[0])\n        sorted_slices = [s for _, s in sorted_pairs]\n        \n        # Stack into volume\n        volume = np.stack([s.pixel_array for s in sorted_slices], axis=-1)\n        \n        # Apply rescale slope/intercept if present\n        ds = sorted_slices[0]\n        if hasattr(ds, 'RescaleSlope') and hasattr(ds, 'RescaleIntercept'):\n            volume = volume * float(ds.RescaleSlope) + float(ds.RescaleIntercept)\n        \n        volume = volume.astype(np.int16)\n        \n        # Get spacing\n        pixel_spacing = ds.PixelSpacing\n        slice_thickness = abs(sorted_pairs[1][0] - sorted_pairs[0][0]) if len(sorted_pairs) > 1 else float(ds.SliceThickness)\n        \n        # Build affine\n        affine = np.diag([\n            float(pixel_spacing[0]),\n            float(pixel_spacing[1]),\n            slice_thickness,\n            1.0\n        ])\n        \n        case_id = f'cervical_{case_idx:04d}'\n        \n        # Save CT image\n        ct_nib = nib.Nifti1Image(volume, affine)\n        nib.save(ct_nib, str(images_dir / f'{case_id}_0000.nii.gz'))\n        \n        # Save label — need to match shape/orientation\n        # Resize mask to match CT volume shape if needed\n        if cervical_mask.shape != volume.shape:\n            from scipy.ndimage import zoom\n            zoom_factors = [\n                volume.shape[0] / cervical_mask.shape[0],\n                volume.shape[1] / cervical_mask.shape[1],\n                volume.shape[2] / cervical_mask.shape[2],\n            ]\n            cervical_mask = zoom(cervical_mask, zoom_factors, order=0).astype(np.uint8)\n        \n        label_nib = nib.Nifti1Image(cervical_mask, affine)\n        nib.save(label_nib, str(labels_dir / f'{case_id}.nii.gz'))\n        \n        print(f\"Saved: {images_dir / f'{case_id}_0000.nii.gz'}\")\n        print(f\"Exists: {os.path.exists(str(images_dir / f'{case_id}_0000.nii.gz'))}\")\n        case_idx += 1\n        \n    except Exception as e:\n        print(f\"Error on {uid}: {e}\")\n        skipped += 1\n\nprint(f\"\\nDone. Converted {case_idx} cases, skipped {skipped}\")\n\n# Write dataset.json\nimport json\ndataset_json = {\n    \"channel_names\": {\"0\": \"CT\"},\n    \"labels\": {\n        \"background\": 0,\n        \"C1\": 1, \"C2\": 2, \"C3\": 3, \"C4\": 4,\n        \"C5\": 5, \"C6\": 6, \"C7\": 7,\n        \"T1\": 8, \"T2\": 9, \"T3\": 10, \"T4\": 11\n    },\n    \"numTraining\": case_idx,\n    \"file_ending\": \".nii.gz\"\n}\nwith open(str(output_dir / 'dataset.json'), 'w') as f:\n    json.dump(dataset_json, f, indent=2)\n\nprint(\"dataset.json written\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-27T13:49:49.719851Z","iopub.execute_input":"2026-03-27T13:49:49.720118Z","iopub.status.idle":"2026-03-27T14:28:19.4648Z","shell.execute_reply.started":"2026-03-27T13:49:49.720086Z","shell.execute_reply":"2026-03-27T14:28:19.464024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Find all nii.gz files in working directory\nfor root, dirs, files in os.walk('/kaggle/working'):\n    for f in files:\n        fullpath = os.path.join(root, f)\n        print(fullpath)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-27T14:28:19.466743Z","iopub.execute_input":"2026-03-27T14:28:19.467048Z","iopub.status.idle":"2026-03-27T14:28:19.473019Z","shell.execute_reply.started":"2026-03-27T14:28:19.46702Z","shell.execute_reply":"2026-03-27T14:28:19.472365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport shutil\nimport os\n\n# Zip the converted dataset\nshutil.make_archive(\n    '/kaggle/working/cervical_nnunet',\n    'zip',\n    '/kaggle/working/',\n    'cervical_nnunet'\n)\nprint(\"Zipped successfully\")\nprint(f\"Size: {os.path.getsize('/kaggle/working/cervical_nnunet.zip') / 1e9:.2f} GB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-27T14:28:19.47403Z","iopub.execute_input":"2026-03-27T14:28:19.47438Z","iopub.status.idle":"2026-03-27T14:34:10.470101Z","shell.execute_reply.started":"2026-03-27T14:28:19.474346Z","shell.execute_reply":"2026-03-27T14:34:10.469369Z"}},"outputs":[],"execution_count":null}]}