{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 0. Introduction\nWelcome to the competition, '<a href=\"https://www.kaggle.com/competitions/rsna-2023-abdominal-trauma-detection/overview\">RSNA 2023 Abdominal Trauma Detection</a>'.  \nAlso, welcome to this source code.  \nThis source code is constructed for the following goals.  \n\n* Providing the function to aggregate separated DICOM (DCM) into DICOM volume (3D tensor, NumPy array).  \n* Visualization function to confirm the slice of converted volume. \n\nTry this source code and upvote if you like it!  \nI hope always good luck to you.  ","metadata":{}},{"cell_type":"markdown","source":"# 1. Preparation\nBefore converting the given DICOM dataset, we will prepare some of the python packages and define some of the python custom functions.  \nNote that, DICOM is a format for storing medical data, and NumPy is a format for storing arrays in Python.  \n\n* DICOM: https://www.dicomstandard.org/  \n* Numpy: https://numpy.org/  ","metadata":{}},{"cell_type":"code","source":"\"\"\" Step 1.\nImport the python packages \"\"\"\n\nimport os, glob, shutil, pickle\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt # for 2D plot\n\nfrom skimage.transform import resize\nfrom tqdm import tqdm # for confirming loop progress\nfrom pydicom import dcmread # for reading the given dicom data\nfrom skimage import measure # for 3D plot","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:29:35.260491Z","iopub.execute_input":"2023-07-28T10:29:35.26089Z","iopub.status.idle":"2023-07-28T10:29:36.168082Z","shell.execute_reply.started":"2023-07-28T10:29:35.260858Z","shell.execute_reply":"2023-07-28T10:29:36.167001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\" Step 2.\nDefine the custom functions for achieving the goal. \"\"\"\n\ndef sorted_list(path): \n    \n    \"\"\" function for getting list of files or directories. \"\"\"\n    \n    tmplist = glob.glob(path) # finding all files or directories and listing them.\n    tmplist.sort() # sorting the found list\n    \n    return tmplist\n\ndef make_dir(path, refresh=False):\n    \n    \"\"\" function for making directory (to save results). \"\"\"\n    \n    try: os.mkdir(path)\n    except: \n        if(refresh): \n            shutil.rmtree(path)\n            os.mkdir(path)\n    \ndef dicom_volume(path, verbose=False, ratio=0): \n    \n    \"\"\" function for getting DICOM volumes and convert to numpy array. \"\"\"\n    \n    list_dcm = sorted_list(path=os.path.join(path, '*.dcm')) # getting all slice as a list\n    list_index = []\n    for path_dcm in list_dcm:\n        list_index.append(int(path_dcm.split('/')[-1].replace('.dcm', ''))) # parsing and adding the index of slice\n    list_index.sort() # sort the index\n    \n    list_arr = [] # array storage\n    for idx_dcm in list_index:\n        if(verbose): print(os.path.join(path, '%d.dcm' %(idx_dcm)))\n        ds = dcmread(os.path.join(path, '%d.dcm' %(idx_dcm))) # getting slice information via single DICOM file.\n        arr = ds.pixel_array # extracting numpy array from DICOM file\n        if(ratio): arr = resize(arr, (arr.shape[0]//ratio, arr.shape[1]//ratio))\n        list_arr.append(arr) # stacking to the array storage\n    \n    return np.asarray(list_arr) # converting as numpy array\n\ndef plot_2d(volume, index=0, title=''): \n    \n    \"\"\" function for plotting the DICOM slice. \"\"\"\n    \n    fig = plt.figure(figsize=(10, 10))\n    ax = fig.add_subplot(111)\n    ax.set_title(title)\n    ax.imshow(volume[index, :, :], cmap='gray')\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:29:36.169817Z","iopub.execute_input":"2023-07-28T10:29:36.170119Z","iopub.status.idle":"2023-07-28T10:29:36.184016Z","shell.execute_reply.started":"2023-07-28T10:29:36.170092Z","shell.execute_reply":"2023-07-28T10:29:36.182952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Conversion\nWe have finished to convert the given DICOM dataset to Numpy array.  \nNow, take a aggregation process.  ","metadata":{}},{"cell_type":"code","source":"\"\"\" Step 1.\nTake a look the list of files given in this competition. \"\"\"\n\nsorted_list(path=os.path.join('../input/rsna-2023-abdominal-trauma-detection', '*'))","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:29:36.185221Z","iopub.execute_input":"2023-07-28T10:29:36.186283Z","iopub.status.idle":"2023-07-28T10:29:36.20562Z","shell.execute_reply.started":"2023-07-28T10:29:36.18625Z","shell.execute_reply":"2023-07-28T10:29:36.204472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\" Step 2.\nThe CSV file will not be DICOM data.\nThus, we need to search inside of the train or test directory. \"\"\"\n\nsorted_list(path=os.path.join('../input/rsna-2023-abdominal-trauma-detection/train_images', '*'))[:10]","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:29:36.208799Z","iopub.execute_input":"2023-07-28T10:29:36.209147Z","iopub.status.idle":"2023-07-28T10:29:36.313275Z","shell.execute_reply.started":"2023-07-28T10:29:36.209118Z","shell.execute_reply":"2023-07-28T10:29:36.31224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\" Step 3.\nLook inside a single ID that is supposed to be a DICOM sample. \nFour types of directory are shown. \"\"\"\n\nsorted_list(path=os.path.join('../input/rsna-2023-abdominal-trauma-detection/train_images/10004', '*'))","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:29:36.314926Z","iopub.execute_input":"2023-07-28T10:29:36.315267Z","iopub.status.idle":"2023-07-28T10:29:36.324966Z","shell.execute_reply.started":"2023-07-28T10:29:36.315239Z","shell.execute_reply":"2023-07-28T10:29:36.323783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\" Step 4.\nLook inside once more a single DICOM volume. \"\"\"\n\nsorted_list(path=os.path.join('../input/rsna-2023-abdominal-trauma-detection/train_images/10004/21057', '*'))[:10]","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:29:36.32615Z","iopub.execute_input":"2023-07-28T10:29:36.326503Z","iopub.status.idle":"2023-07-28T10:29:36.988214Z","shell.execute_reply.started":"2023-07-28T10:29:36.326473Z","shell.execute_reply":"2023-07-28T10:29:36.98734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\" Step 5\nNow, we can convert each type of DICOM for each ID. \"\"\"\n\ntmp_vol = dicom_volume(path='../input/rsna-2023-abdominal-trauma-detection/train_images/10004/21057')\n\nprint(\"Converted Array\")\nprint(tmp_vol.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:29:36.989201Z","iopub.execute_input":"2023-07-28T10:29:36.98954Z","iopub.status.idle":"2023-07-28T10:30:00.242264Z","shell.execute_reply.started":"2023-07-28T10:29:36.989512Z","shell.execute_reply":"2023-07-28T10:30:00.240968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Visualization\nWe conduct the visualization task to confirm the DICOM file has been converted as numpy correctly.  ","metadata":{}},{"cell_type":"code","source":"\"\"\" Step 1\n2D plot for each DICOM type. \"\"\"\n\nplot_2d(tmp_vol, index=0, title=\"Volume\")\nplot_2d(tmp_vol, index=int(tmp_vol.shape[0]/2), title=\"Volume\")\nplot_2d(tmp_vol, index=-1, title=\"Volume\")","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:30:00.243938Z","iopub.execute_input":"2023-07-28T10:30:00.244953Z","iopub.status.idle":"2023-07-28T10:30:01.927761Z","shell.execute_reply.started":"2023-07-28T10:30:00.244897Z","shell.execute_reply":"2023-07-28T10:30:01.926607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Full Conversion\nWe can get fully converted dataset via following procedure.","metadata":{}},{"cell_type":"code","source":"\"\"\" Step 1\nBefore sore the converted DICOM volume, load labels.\nLabel information will be store with corresponding DICOM volume. \"\"\"\n\ndf_image_label = pd.read_csv('../input/rsna-2023-abdominal-trauma-detection/image_level_labels.csv')\ndf_image_label.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:30:01.929104Z","iopub.execute_input":"2023-07-28T10:30:01.929471Z","iopub.status.idle":"2023-07-28T10:30:01.978759Z","shell.execute_reply.started":"2023-07-28T10:30:01.929428Z","shell.execute_reply":"2023-07-28T10:30:01.977461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\" Step 2\nDefine functions for saving and loading. \"\"\"\n\ndef save_pkl(path, pkl):\n\n    with open(path,'wb') as fw:\n        pickle.dump(pkl, fw)\n\ndef load_pkl(path):\n\n    with open(path, 'rb') as fr:\n        pkl = pickle.load(fr)\n\n    return pkl","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:30:01.982925Z","iopub.execute_input":"2023-07-28T10:30:01.983296Z","iopub.status.idle":"2023-07-28T10:30:01.990664Z","shell.execute_reply.started":"2023-07-28T10:30:01.983266Z","shell.execute_reply":"2023-07-28T10:30:01.989033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\" Step 3\nConvert DICOMs into NumPy array. \nThen, store them with label information. \nDue to the image size being large, each horizontal and vertical resolution is reduced to 1/4 before saving (the channel axis is not reduced). \"\"\"\n\nsave_root = 'converted'\nmake_dir(path=save_root, refresh=True)\n\nfor category in ['train', 'test']:\n    make_dir(path=os.path.join(save_root, category), refresh=False)\n    list_patient_id = sorted_list(path='../input/rsna-2023-abdominal-trauma-detection/%s_images/*' %(category))\n    \n    print(\"Convert %s-set\" %(category))\n    for idx_patent, path_patient_id in tqdm(enumerate(list_patient_id)):\n        patient_id = path_patient_id.split('/')[-1]\n        make_dir(path=os.path.join(save_root, category, patient_id), refresh=False)\n        list_series_id = sorted_list(path='%s/*' %(path_patient_id))\n\n        for path_series_id in list_series_id:\n            series_id = path_series_id.split('/')[-1]\n            tmp_volume = dicom_volume(path=path_series_id, ratio=4)\n            save_path = os.path.join(save_root, category, patient_id, '%s.pkl' %(series_id))\n            \n            try: \n                tmp_label = df_image_label.loc[(df_image_label['patient_id'] == int(patient_id)) & (df_image_label['series_id'] == int(series_id))]\n            except:\n                tmp_label = None\n            save_pkl(save_path, {'volume':tmp_volume, 'label':tmp_label})","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:30:01.992278Z","iopub.execute_input":"2023-07-28T10:30:01.992686Z","iopub.status.idle":"2023-07-28T10:31:08.709928Z","shell.execute_reply.started":"2023-07-28T10:30:01.992654Z","shell.execute_reply":"2023-07-28T10:31:08.708814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Conversion Confirmation\nWe will confirm the saved DICOM volumes in this section.","metadata":{}},{"cell_type":"code","source":"sorted_list(path='../working/converted/*/*/*.pkl')","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:31:08.711889Z","iopub.execute_input":"2023-07-28T10:31:08.713088Z","iopub.status.idle":"2023-07-28T10:31:08.812891Z","shell.execute_reply.started":"2023-07-28T10:31:08.713044Z","shell.execute_reply":"2023-07-28T10:31:08.811681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_pkl = load_pkl('converted/train/10004/21057.pkl')","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:31:08.814522Z","iopub.execute_input":"2023-07-28T10:31:08.814874Z","iopub.status.idle":"2023-07-28T10:31:08.902903Z","shell.execute_reply.started":"2023-07-28T10:31:08.814845Z","shell.execute_reply":"2023-07-28T10:31:08.901856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_2d(tmp_pkl['volume'], index=0, title=\"Volume\")\nplot_2d(tmp_pkl['volume'], index=int(tmp_pkl['volume'].shape[0]/2), title=\"Volume\")\nplot_2d(tmp_pkl['volume'], index=-1, title=\"Volume\")","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:31:08.90436Z","iopub.execute_input":"2023-07-28T10:31:08.904879Z","iopub.status.idle":"2023-07-28T10:31:10.072724Z","shell.execute_reply.started":"2023-07-28T10:31:08.904837Z","shell.execute_reply":"2023-07-28T10:31:10.071378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_pkl['label']","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:31:10.074815Z","iopub.execute_input":"2023-07-28T10:31:10.075171Z","iopub.status.idle":"2023-07-28T10:31:10.094031Z","shell.execute_reply.started":"2023-07-28T10:31:10.075143Z","shell.execute_reply":"2023-07-28T10:31:10.092706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}