{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import necessary libraries\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport pydicom\nimport glob\n# Step 1: Load the Data\ntrain_df = pd.read_csv('/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv')\ntest_df = pd.read_csv('/kaggle/input/rsna-2022-cervical-spine-fracture-detection/test.csv')\ntrain_bboxes_df = pd.read_csv('/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_bounding_boxes.csv')\n\n# Step 2: Explore Target Variables\ndef plot_target_distribution(df, target_column):\n    plt.figure(figsize=(8, 6))\n    df[target_column].value_counts().plot(kind='bar', color='skyblue')\n    plt.title(f'Distribution of {target_column}')\n    plt.xlabel(target_column)\n    plt.ylabel('Count')\n    plt.show()\n\nplot_target_distribution(train_df, 'patient_overall')\nfor vertebra in ['C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']:\n    plot_target_distribution(train_df, vertebra)\n\n# Step 3: Study the Images\ndef load_and_display_dicom_image(image_path):\n    dcm_data = pydicom.dcmread(image_path)\n    plt.imshow(dcm_data.pixel_array, cmap=plt.cm.bone)\n    plt.title('Sample DICOM Image')\n    plt.axis('off')\n    plt.show()\n\n# Load and display a few DICOM images\nsample_images_dir = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images'  # Replace with the correct path to the train_images directory\npatient_dirs = os.listdir(sample_images_dir)\n\nfor patient_dir in patient_dirs[:3]:\n    # Use glob to retrieve all DICOM files inside each patient directory\n    patient_image_files = glob.glob(os.path.join(sample_images_dir, patient_dir, '*.dcm'))\n    \n    # Display the first DICOM image for each patient\n    if patient_image_files:\n        load_and_display_dicom_image(patient_image_files[0])\n\n# Step 4: Investigate Bounding Boxes (if available)\n# You can plot bounding boxes on images using the train_bboxes_df data.\n# (Same code as provided earlier)\n\n# Step 5: Analyze Segmentations (if available)\ndef load_and_display_nifti_segmentation(segmentation_path):\n    nifti_data = nib.load(segmentation_path)\n    plt.imshow(nifti_data.get_fdata()[:, :, 0], cmap=plt.cm.bone)\n    plt.title('Sample NIFTI Segmentation')\n    plt.axis('off')\n    plt.show()\n\n# Load and display a few NIFTI segmentations\nsample_segmentations_dir = 'segmentations/'  # Replace with the correct path to the segmentations directory\nsample_segmentation_files = glob.glob(os.path.join(sample_segmentations_dir, '*.nii'))[:3]\nfor segmentation_file in sample_segmentation_files:\n    load_and_display_nifti_segmentation(segmentation_file)\n\n\n\n\n\n\n\n\n\n\n\n# Deep Learning / Programming:\n# For deep learning tasks, you can preprocess the images, perform data augmentation, and prepare the data for training deep learning models. You can also implement and train neural networks, such as Convolutional Neural Networks (CNNs), to predict cervical spine fractures based on the images. Consider using libraries like TensorFlow or PyTorch for deep learning model development.\n\n# Remember to document your code thoroughly, use appropri","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-25T20:00:15.814633Z","iopub.execute_input":"2023-07-25T20:00:15.815072Z","iopub.status.idle":"2023-07-25T20:00:18.757647Z","shell.execute_reply.started":"2023-07-25T20:00:15.815039Z","shell.execute_reply":"2023-07-25T20:00:18.756747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 6: Check Data Imbalance\n# Calculate the number of fractured vs. non-fractured cases in the target variables.\n\ndef calculate_target_counts(df, target_column):\n    target_counts = df[target_column].value_counts()\n    return target_counts\n\n# Calculate the counts for 'patient_overall'\npatient_overall_counts = calculate_target_counts(train_df, 'patient_overall')\nprint(\"Patient Overall Target Counts:\")\nprint(patient_overall_counts)\n\n# Calculate the counts for each vertebra from C1 to C7\nvertebrae = ['C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']\nfor vertebra in vertebrae:\n    vertebra_counts = calculate_target_counts(train_df, vertebra)\n    print(f\"{vertebra} Target Counts:\")\n    print(vertebra_counts)\n\n# Step 6: Visualize Data Imbalance (optional)\ndef plot_target_counts(target_counts, target_column):\n    plt.figure(figsize=(8, 6))\n    target_counts.plot(kind='bar', color='skyblue')\n    plt.title(f'Data Imbalance for {target_column}')\n    plt.xlabel(target_column)\n    plt.ylabel('Count')\n    plt.show()\n\n# Plot the data imbalance for 'patient_overall'\nplot_target_counts(patient_overall_counts, 'patient_overall')\n\n# Plot the data imbalance for each vertebra from C1 to C7\nfor vertebra in vertebrae:\n    vertebra_counts = calculate_target_counts(train_df, vertebra)\n    plot_target_counts(vertebra_counts, vertebra)","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:00:18.759753Z","iopub.execute_input":"2023-07-25T20:00:18.760492Z","iopub.status.idle":"2023-07-25T20:00:20.797635Z","shell.execute_reply.started":"2023-07-25T20:00:18.760456Z","shell.execute_reply":"2023-07-25T20:00:20.796712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 7: Handle Missing Data (if any)\n\n# Check for missing data in train_df\ntrain_missing_data = train_df.isnull().sum()\nprint(\"Missing Data in train_df:\")\nprint(train_missing_data)\n\n# Check for missing data in test_df\ntest_missing_data = test_df.isnull().sum()\nprint(\"Missing Data in test_df:\")\nprint(test_missing_data)\n\n# After handling missing data, you can recheck to ensure that there are no more missing values:\nprint(\"Missing Data After Handling:\")\nprint(train_df.isnull().sum())\nprint(test_df.isnull().sum())\n# there is no missing data in both the 'train_df' and 'test_df' DataFrames","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:00:20.799079Z","iopub.execute_input":"2023-07-25T20:00:20.800109Z","iopub.status.idle":"2023-07-25T20:00:20.815046Z","shell.execute_reply.started":"2023-07-25T20:00:20.800074Z","shell.execute_reply":"2023-07-25T20:00:20.813924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n# Step 8: Explore Relationships\n# Step 8: Explore Relationships\n# Bar Plot\ndef create_bar_plot(x, y, title):\n    plt.figure(figsize=(8, 6))\n    sns.barplot(x=x, y=y, data=train_df)\n    plt.title(title)\n    plt.xlabel(x)\n    plt.ylabel(y)\n    plt.show()\n\n# Choose two columns to create a bar plot\n# For example, let's create a bar plot between 'C1' and 'C2'\ncreate_bar_plot(x='C1', y='C2', title='Bar Plot: C1 vs. C2')\n# Scatter Plot\ndef create_scatter_plot(x, y, title):\n    plt.figure(figsize=(8, 6))\n    sns.scatterplot(x=x, y=y, data=train_df)\n    plt.title(title)\n    plt.xlabel(x)\n    plt.ylabel(y)\n    plt.show()\n\n# Choose two columns to create a scatter plot\n# For example, let's create a scatter plot between 'C1' and 'C2'\ncreate_scatter_plot(x='C1', y='C2', title='Scatter Plot: C1 vs. C2')\n\n# Box Plot\ndef create_box_plot(x, y, title):\n    plt.figure(figsize=(8, 6))\n    sns.boxplot(x=x, y=y, data=train_df)\n    plt.title(title)\n    plt.xlabel(x)\n    plt.ylabel(y)\n    plt.show()\n\n# Choose two columns to create a box plot\n# For example, let's create a box plot between 'patient_overall' and 'C1'\ncreate_box_plot(x='patient_overall', y='C1', title='Box Plot: Patient Overall vs. C1')\n\n# Correlation Matrix\ndef create_correlation_matrix(df):\n    correlation_matrix = df.corr()\n    plt.figure(figsize=(10, 8))\n    sns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', fmt=\".2f\")\n    plt.title('Correlation Matrix')\n    plt.show()\n\n# Create a correlation matrix for selected numerical columns in the DataFrame\n# Note: 'StudyInstanceUID' is a unique identifier and should be excluded from the correlation matrix\nnumerical_columns = train_df.select_dtypes(include='number').columns.tolist()\nnumerical_columns = [col for col in numerical_columns if col != 'StudyInstanceUID']\ncreate_correlation_matrix(train_df[numerical_columns])","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:00:20.817907Z","iopub.execute_input":"2023-07-25T20:00:20.818306Z","iopub.status.idle":"2023-07-25T20:00:22.185752Z","shell.execute_reply.started":"2023-07-25T20:00:20.818274Z","shell.execute_reply":"2023-07-25T20:00:22.184788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport cv2\n\n# Step 2: Explore Data\nprint(\"Training Data Info:\")\nprint(train_df.info())\n\nprint(\"Test Data Info:\")\nprint(test_df.info())\n\nprint(\"Train Bounding Boxes Info:\")\nprint(train_bboxes_df.info())\n\nfor col in ['C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']:\n    print(f\"{col} Target Counts:\")\n    print(train_df[col].value_counts())\n\n# Step 4: Handle Missing Data (if any)\nprint(\"Missing Data in train_df:\")\nprint(train_df.isnull().sum())\n\nprint(\"Missing Data in test_df:\")\nprint(test_df.isnull().sum())\n\n# Assuming the train_image_dir is provided correctly\ntrain_image_dir = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images'\n\"\"\"\n# Step 5: Get Image Paths\ndef get_image_paths(image_dir):\n    image_paths = []\n    for root, _, files in os.walk(image_dir):\n        for file in files:\n            if file.endswith('.dcm'):\n                image_paths.append(os.path.join(root, file))\n    return image_paths\n\nimage_paths = get_image_paths(train_image_dir)\n\n# Step 6:# Assuming the train_image_dir is provided correctly\ntrain_image_dir = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images'\n\n# Step 5: Get Image Paths\ndef get_image_paths(image_dir):\n    image_paths = []\n    for root, _, files in os.walk(image_dir):\n        for file in files:\n            if file.endswith('.dcm'):\n                image_paths.append(os.path.join(root, file))\n    return image_paths\n\n# Get Image Paths Header\nimage_paths_header = [\"Image Paths\"]\nimage_paths_df = pd.DataFrame(get_image_paths(train_image_dir), columns=image_paths_header)\nprint(\"Image Paths:\")\nprint(image_paths_df.head())\ndef get_image_info(image_paths):\n    image_info = []\n    for path in image_paths:\n        image = cv2.imread(path)\n        if image is not None:\n            height, width, _ = image.shape\n            format_ = path.split('.')[-1]\n            image_info.append({'Path': path, 'Height': height, 'Width': width, 'Format': format_})\n        else:\n            print(f\"Warning: Unable to read image at path {path}\")\n    return pd.DataFrame(image_info)\n\nimage_info_df = get_image_info(image_paths)\nprint(\"Image Dimensions and Formats:\")\nprint(image_info_df.head())\n\n\n\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:00:22.187361Z","iopub.execute_input":"2023-07-25T20:00:22.188024Z","iopub.status.idle":"2023-07-25T20:00:22.229069Z","shell.execute_reply.started":"2023-07-25T20:00:22.187988Z","shell.execute_reply":"2023-07-25T20:00:22.228054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 7: Explore Relationships (using pair plot)\ndef create_pair_plot(df):\n    sns.pairplot(df, hue='patient_overall', diag_kind='kde')\n    plt.show()\n\ncreate_pair_plot(train_df)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:00:22.230541Z","iopub.execute_input":"2023-07-25T20:00:22.232316Z","iopub.status.idle":"2023-07-25T20:00:44.908887Z","shell.execute_reply.started":"2023-07-25T20:00:22.23228Z","shell.execute_reply":"2023-07-25T20:00:44.907791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport numpy as np\nimport cv2\n\ndef resize_normalize_images(image_paths, target_size):\n    normalized_images = []\n    for path in image_paths:\n        # Read the image using PIL\n        image = Image.open(path)\n        # Resize the image to the target size\n        image = image.resize(target_size)\n        # Convert the image to an array\n        image_array = cv2.cvtColor(np.array(image), cv2.COLOR_RGB2BGR)\n        \n        image = image.convert('RGB')\n\n        # Normalize pixel values to the range [0, 1]\n        normalized_image = image_array / 255.0\n        normalized_images.append(normalized_image)\n    return normalized_images\n","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:00:44.910391Z","iopub.execute_input":"2023-07-25T20:00:44.910907Z","iopub.status.idle":"2023-07-25T20:00:44.919094Z","shell.execute_reply.started":"2023-07-25T20:00:44.910871Z","shell.execute_reply":"2023-07-25T20:00:44.91791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Step 9: Optionally, visualize the 3D bounding boxes on the original medical images for verification\n# (visualization code not included here)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:00:44.920736Z","iopub.execute_input":"2023-07-25T20:00:44.92109Z","iopub.status.idle":"2023-07-25T20:00:44.931628Z","shell.execute_reply.started":"2023-07-25T20:00:44.921057Z","shell.execute_reply":"2023-07-25T20:00:44.930677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:01:08.177288Z","iopub.execute_input":"2023-07-25T20:01:08.177622Z","iopub.status.idle":"2023-07-25T20:01:08.184806Z","shell.execute_reply.started":"2023-07-25T20:01:08.177593Z","shell.execute_reply":"2023-07-25T20:01:08.18389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:01:19.786143Z","iopub.execute_input":"2023-07-25T20:01:19.786547Z","iopub.status.idle":"2023-07-25T20:01:19.799618Z","shell.execute_reply.started":"2023-07-25T20:01:19.786507Z","shell.execute_reply":"2023-07-25T20:01:19.798264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:01:19.866388Z","iopub.status.idle":"2023-07-25T20:01:19.866889Z","shell.execute_reply.started":"2023-07-25T20:01:19.866624Z","shell.execute_reply":"2023-07-25T20:01:19.866647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:01:19.869669Z","iopub.status.idle":"2023-07-25T20:01:19.870913Z","shell.execute_reply.started":"2023-07-25T20:01:19.870592Z","shell.execute_reply":"2023-07-25T20:01:19.87062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-07-25T20:01:19.872107Z","iopub.status.idle":"2023-07-25T20:01:19.872748Z","shell.execute_reply.started":"2023-07-25T20:01:19.872499Z","shell.execute_reply":"2023-07-25T20:01:19.872523Z"},"trusted":true},"execution_count":null,"outputs":[]}]}