{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# load the libraries\nlibrary(tidyverse) # metapackage of all tidyverse packages\nlibrary(skimr) # for quick examination of the data\nlibrary(ggthemes) # for themes\nlibrary(oro.nifti) # for reading .nii files\nlibrary(scales) # for scaling color of .nii files\nlibrary(oro.dicom) # for viewing DICOM files\nlibrary(reticulate) # for loading python libraries\nlibrary(arrow) # for reading parquet files\nlibrary(shades) # used for image shading\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-31T19:02:17.69864Z","iopub.execute_input":"2023-08-31T19:02:17.700343Z","iopub.status.idle":"2023-08-31T19:02:19.918696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the .csv data\ntrain <- read.csv('../input/rsna-2023-abdominal-trauma-detection/train.csv')\ntrain_series_meta <- read.csv('../input/rsna-2023-abdominal-trauma-detection/train_series_meta.csv')\nimage_level_labels <- read.csv('../input/rsna-2023-abdominal-trauma-detection/image_level_labels.csv')\n\n# Read parquet file\noptions(arrow.skip_nul = TRUE) # required to deal with null values, otherwise errors out during parquet read\nparq_train <- read_parquet('../input/rsna-2023-abdominal-trauma-detection/train_dicom_tags.parquet', as_data_frame = TRUE) # read train parquet\nparq_test <- read_parquet('../input/rsna-2023-abdominal-trauma-detection/test_dicom_tags.parquet', as_data_frame = TRUE) # read test parquet\n","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:02:19.921235Z","iopub.execute_input":"2023-08-31T19:02:19.953544Z","iopub.status.idle":"2023-08-31T19:02:21.174329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Information on the total records in the parquet data file\nparq_train %>% nrow()","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:02:21.177573Z","iopub.execute_input":"2023-08-31T19:02:21.178689Z","iopub.status.idle":"2023-08-31T19:02:21.195635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parse out series_id from 'path' in the parquet data set; to be used later on.\n# Need to create a copy of path because 'separate' will consume that column, additionally need to discard unused/redundant portions of path\n\nparq_train <- parq_train %>% separate(path, into = c('throw1', 'throw2', 'throw3', 'series_id', 'throw4'), remove=FALSE) %>% select(-throw1,-throw2,-throw3,-throw4)\nparq_test <- parq_test %>% separate(path, into = c('throw1', 'throw2', 'throw3', 'series_id', 'throw4'), remove=FALSE) %>% select(-throw1,-throw2,-throw3,-throw4)","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:02:21.199052Z","iopub.execute_input":"2023-08-31T19:02:21.200175Z","iopub.status.idle":"2023-08-31T19:02:53.802817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Information on the Frame of Reference UID\n# From Reference # 2, the Frame of Reference, each series has only one Frame of Reference UID, but series can share a Frame of Reference UID.\n# There are 3147 unique Frame of Reference UIDs\n\nparq_train %>% select(FrameOfReferenceUID) %>% unique()","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:02:53.806093Z","iopub.execute_input":"2023-08-31T19:02:53.807279Z","iopub.status.idle":"2023-08-31T19:02:56.39666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Information about the Image Position Patient\n# Mostly unique; 1426166 unique values out of 1510373\n\nparq_train %>% select(ImagePositionPatient) %>% unique()","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:02:56.399709Z","iopub.execute_input":"2023-08-31T19:02:56.400985Z","iopub.status.idle":"2023-08-31T19:02:58.698019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Information about the Image Orientation Patient\n# Unique values show 173; however some appear to be differences with formatting.\n\n#parq %>% select(ImageOrientationPatient) %>% unique()\nparq_train %>% group_by(ImageOrientationPatient) %>% summarize(totals = n())","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:02:58.701303Z","iopub.execute_input":"2023-08-31T19:02:58.702511Z","iopub.status.idle":"2023-08-31T19:02:58.910969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Information about the PhotometricInterpretation\n# Only one unique value: MONOCHROME2\n\nparq_train %>% group_by(PhotometricInterpretation) %>% summarize(total = n())","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:02:58.914178Z","iopub.execute_input":"2023-08-31T19:02:58.915554Z","iopub.status.idle":"2023-08-31T19:02:59.000001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Information about the PatientPosition.\n# Only two unique values: FFS, HFS\n# From Reference # 2 below, these refer to Head First Supine & Feet First Supine. \n\nparq_train %>% group_by(PatientPosition) %>% summarize(total = n())","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:02:59.003351Z","iopub.execute_input":"2023-08-31T19:02:59.004732Z","iopub.status.idle":"2023-08-31T19:02:59.0918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Information about the WindowCenter & WindowWidths\n# There is only one unique combination: Center: 50, Width: 400\n# From information obtained from Reference # 1 below, this is related to soft tissues.\n\nparq_train %>% select(WindowCenter, WindowWidth) %>% unique()","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:02:59.095061Z","iopub.execute_input":"2023-08-31T19:02:59.096261Z","iopub.status.idle":"2023-08-31T19:03:01.494968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the training images for this competition\ntrain_images <- list.files(path = '../input/rsna-2023-abdominal-trauma-detection/train_images', full.names = TRUE, recursive = TRUE)\n\n# Make it a dataframe \ntrain_images <- data.frame(path = train_images)\n\n# Parse out patient_id, series_id, & instance_number for easier access later on to images by those variables & for comparison of unique patient_ids\ntrain_images <- train_images %>% separate(path, into = c('throw1','throw2','throw3','throw4','patient_id','series_id','instancenumber'), sep='/', remove=FALSE) %>% \nseparate(instancenumber, into = c('instance_number', 'throw5')) %>% select(-throw1,-throw2,-throw3,-throw4, -throw5)","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:03:01.497903Z","iopub.execute_input":"2023-08-31T19:03:01.498985Z","iopub.status.idle":"2023-08-31T19:10:16.876176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of images to process from the list of images\nnrow(train_images)\n\n# Why is this number different than the number of records in the parquet data, which was 1510373?","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:10:16.878141Z","iopub.execute_input":"2023-08-31T19:10:16.879184Z","iopub.status.idle":"2023-08-31T19:10:16.892188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Unique Patient_IDs in the various datasets\nparq_train %>% select(PatientID) %>% unique() %>% nrow()\ntrain %>% select(patient_id) %>% unique() %>% nrow()\ntrain_series_meta %>% select(patient_id) %>% unique() %>% nrow()\ntrain_images %>% select(patient_id) %>% unique() %>% nrow()\nimage_level_labels %>% select(patient_id) %>% unique() %>% nrow()\n# Why are there only 246 unique patient_ids in the image_level_labels.csv?","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:10:16.894088Z","iopub.execute_input":"2023-08-31T19:10:16.895091Z","iopub.status.idle":"2023-08-31T19:10:16.931249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use this section to explore what the image_level_labels.csv does \n\nimage_level_labels %>% select(injury_name) %>% unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Percent complete, incomplete in train_series_meta.csv\ntrain_series_meta %>% summarize(incomplete = sum(incomplete_organ), complete = n()-sum(incomplete_organ)) %>% \npivot_longer(cols = everything()) %>% ggplot(aes(x=name, y=value)) + geom_col() + \nxlab('CT organ status') + ylab('Count') + ggtitle('CT organ status in train_series_meta.csv') + theme_clean()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Percent healthy, any_injury in train.csv\ntrain %>% summarize(Healthy = (n() - sum(any_injury))/n(), Any_Injury = sum(any_injury)/n()) %>% pivot_longer(cols = everything()) %>% \nggplot(aes(x=reorder(name,-value), y=value)) + geom_col() + \nxlab('Status') + ylab('Count') + ggtitle('Healthy vs. Any Injury Status in train.csv') + theme_clean()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Individual injuries as a percentage of any_injury in train.csv\ntrain %>% reframe(bowel_injured_percent = sum(bowel_injury)/sum(any_injury), \n                  extravassation_injured_percent = sum(extravasation_injury)/sum(any_injury), \n                  kidney_low_percent = sum(kidney_low)/sum(any_injury), \n                  kidney_high_percent = sum(kidney_high)/sum(any_injury), \n                  liver_low_percent = sum(liver_low)/sum(any_injury),  \n                  liver_high_percent = sum(liver_high)/sum(any_injury), \n                  spleen_low_percent = sum(spleen_low)/sum(any_injury),  \n                  spleen_high_percent = sum(spleen_high)/sum(any_injury)) %>% \npivot_longer(cols = everything()) %>% ggplot(aes(x=reorder(name,value), y=value)) + \ngeom_col() + coord_flip(clip = \"off\") + scale_x_discrete(guide = guide_axis(angle = 0)) + theme_clean() +\nxlab('Injury Type') + ylab('Count') + ggtitle(str_wrap('Individual Injury Percentage of Any Injury in train.csv', 40))","metadata":{"execution":{"iopub.status.busy":"2023-08-31T19:43:04.8516Z","iopub.execute_input":"2023-08-31T19:43:04.85294Z","iopub.status.idle":"2023-08-31T19:43:05.391146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Initial code to download & view images ###","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load python packages via reticulate library \npyfiles1<-c(\"gdcm\",\"pydicom\")\npy_install(pyfiles1)\n\n# import python packages\npydicom <-import(\"pydicom\", convert = FALSE)\nnp <- import(\"numpy\", convert = TRUE)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read a single .dcm file & display\n\n# Read the file, grab the first image from the patientDir list for an example\ndicom = pydicom$read_file(train_images[1,1])\n\n# Plot the image\nimage(np$asmatrix(dicom$pixel_array), col=grey(0:64/64), axes=FALSE, xlab=\"\", ylab=\"\")  ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize incomplete organ images from train_series_meta.csv\n# Because we parsed out patient_id, series_id, and instance_number from the file path in the train images, this is easy to do\n\nincomplete_organ_df <- train_series_meta %>% filter(incomplete_organ == 1) # get list of incomplete organs from the train_series_meta.csv\n\nincomplete_organ_scans <- merge(incomplete_organ_df, train_images, by.x=c('patient_id', 'series_id'), by.y=c('patient_id', 'series_id'), all.x = TRUE) # left join w/ train_images\n\n# Grab the first five and print\nfor(i in 1:5){\n    dicom = pydicom$read_file(incomplete_organ_scans[i,5]) # grab the path to i'th image\n    oro.nifti::image(np$asmatrix(dicom$pixel_array), col=grey(0:64/64), axes=TRUE, xlab=\"\", ylab=\"\",)  # Plot the image\n}\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test a baseline submission using random sampling based on percent healthy vs. any_injury, and injury percentages in the training data\n# Not the best approach!  But want to test submit functionality\n\n# Read the test_series_meta.csv\ntest_series_meta <- read.csv('../input/rsna-2023-abdominal-trauma-detection/test_series_meta.csv')\n\n# Gather unique test patient_ids\nunique_test_patient_ids <- test_series_meta %>% select(patient_id) %>% unique()\n\n# Create the empty submission dataframe which we will add to\nsubmission_df <- setNames(data.frame(matrix(ncol = 14, nrow = 0)), c('patient_id', 'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury', \n                                                                    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high', \n                                                                     'spleen_healthy', 'spleen_low', 'spleen_high'))\nunique_test_patient_ids\nfor(i in 1:nrow(unique_test_patient_ids)){\n    # first determine if healthy or any_injury\n    # in this example, 1=healthy, 0=any_injury\n    # From the train data, we have 72.8% healthy, 27.2% any_injury, so we will use those for our submission\n    \n    health_status <- sample(c(1,0), size=1, prob=c(.728, .272))\n    print(paste('health = ', health_status))\n    if(health_status == 1){\n        #all healthy codes\n        tmp_df <- setNames(data.frame(matrix(ncol = 14, nrow = 0)), c('patient_id', 'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury', \n                                                                    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high', \n                                                                     'spleen_healthy', 'spleen_low', 'spleen_high'))\n        tmp_df[1,] <- c(unique_test_patient_ids[i,1], 1,0,1,0,1,0,0,1,0,0,1,0,0)\n        submission_df <- rbind(submission_df,tmp_df)\n    }\n    else{\n        #any_injury...for now, just provide the inverse of the healthy submission\n        tmp_df <- setNames(data.frame(matrix(ncol = 14, nrow = 0)), c('patient_id', 'bowel_healthy', 'bowel_injury', 'extravasation_healthy', 'extravasation_injury', \n                                                                    'kidney_healthy', 'kidney_low', 'kidney_high', 'liver_healthy', 'liver_low', 'liver_high', \n                                                                     'spleen_healthy', 'spleen_low', 'spleen_high'))\n        tmp_df[1,] <- c(unique_test_patient_ids[i,1], 0,1,0,1,0,1,1,0,1,1,0,1,1)      \n        submission_df <- rbind(submission_df,tmp_df)\n    }\n    \n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"write_csv(submission_df, 'submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# From researching on how to tackle this competition, I know there are some pixel processing / orientation / formatting tasks that need to be done. \n# I plan on posting that when I complete it as well.  \n# Just wanted to provide other R users an initial capability in accessing & viewing the images.\n# I'll continue to update -- please vote if it's helpful or leave a commment to make it better!","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# References -- thanks!\n# (1) https://www.kaggle.com/code/redwankarimsony/ct-scans-dicom-files-windowing-explained \n# (2) https://dicom.innolitics.com\n# (3) https://www.alexejgossmann.com/MRI_viz/","metadata":{"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"}}