{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Introduction:\n\nThis notebook contains various utility code snippets to prepare datasets in the required specs/format.\n\n\n### 1. DICOM to PNG\n\nWe'll convert DICOM images in the training set to PNG images of required sizes. The PNG images will be easier to work with and intially it's helpful to prototype with smaller 128x128 or 256x256 images. Most of this code is based on this [notebook](https://www.kaggle.com/code/theoviel/get-started-quicker-dicom-png-conversion).","metadata":{}},{"cell_type":"code","source":"!pip install -qU python-gdcm pylibjpeg","metadata":{"execution":{"iopub.status.busy":"2023-09-26T09:11:48.568361Z","iopub.execute_input":"2023-09-26T09:11:48.568819Z","iopub.status.idle":"2023-09-26T09:12:02.804264Z","shell.execute_reply.started":"2023-09-26T09:11:48.568787Z","shell.execute_reply":"2023-09-26T09:12:02.802922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\nimport glob\nimport gdcm\nimport pydicom\nimport matplotlib.pyplot as plt\nimport random\nimport shutil\nrandom.seed(123)\n\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-26T09:12:05.988377Z","iopub.execute_input":"2023-09-26T09:12:05.988814Z","iopub.status.idle":"2023-09-26T09:12:05.997656Z","shell.execute_reply.started":"2023-09-26T09:12:05.988779Z","shell.execute_reply":"2023-09-26T09:12:05.996093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DICOM_PATH = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/'\nSAVE_FOLDER = '/kaggle/working/train_images_128_png_complete/'\nSIZE = 128\nEXTENSION = 'png'\n\nos.makedirs(SAVE_FOLDER, exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T10:33:42.485717Z","iopub.execute_input":"2023-09-16T10:33:42.486191Z","iopub.status.idle":"2023-09-16T10:33:42.493658Z","shell.execute_reply.started":"2023-09-16T10:33:42.486153Z","shell.execute_reply":"2023-09-16T10:33:42.492678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def standardize_pixel_array(dcm: pydicom.dataset.FileDataset) -> np.ndarray:\n    \"\"\"\n    Source : https://www.kaggle.com/competitions/rsna-2023-abdominal-trauma-detection/discussion/427217\n    \"\"\"\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    pixel_array = dcm.pixel_array\n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        pixel_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n#         pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n\n    intercept = float(dcm.RescaleIntercept)\n    slope = float(dcm.RescaleSlope)\n    center = int(dcm.WindowCenter)\n    width = int(dcm.WindowWidth)\n    low = center - width / 2\n    high = center + width / 2    \n    \n    pixel_array = (pixel_array * slope) + intercept\n    pixel_array = np.clip(pixel_array, low, high)\n\n    return pixel_array","metadata":{"execution":{"iopub.status.busy":"2023-09-12T10:29:22.412292Z","iopub.execute_input":"2023-09-12T10:29:22.412654Z","iopub.status.idle":"2023-09-12T10:29:22.422621Z","shell.execute_reply.started":"2023-09-12T10:29:22.412622Z","shell.execute_reply":"2023-09-12T10:29:22.42132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(patient, size = SIZE, save_folder = \"\", data_path = \"\"):\n    \n    for study in sorted(os.listdir(data_path + patient)):\n        imgs = {}\n        for f in sorted(glob.glob(data_path + f\"{patient}/{study}/*.dcm\")):\n            \n            dicom = pydicom.dcmread(f)\n            pos_z = dicom[(0x20, 0x32)].value[-1]\n            \n            img = standardize_pixel_array(dicom)\n            img = (img - img.min())/(img.max() - img.min() + 1e-6)\n            \n            if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n                img = 1 - img\n            \n            imgs[pos_z] = img\n        \n        for i, k in enumerate(sorted(imgs.keys())):\n            \n            img = imgs[k]\n            \n            img = cv2.resize(img, (size, size))\n            cv2.imwrite(save_folder + f\"{patient}_{study}_{i}.png\", (img * 255).astype(np.uint8))","metadata":{"execution":{"iopub.status.busy":"2023-09-12T10:29:22.424577Z","iopub.execute_input":"2023-09-12T10:29:22.425122Z","iopub.status.idle":"2023-09-12T10:29:22.436196Z","shell.execute_reply.started":"2023-09-12T10:29:22.425068Z","shell.execute_reply":"2023-09-12T10:29:22.435334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patients = os.listdir(DICOM_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T10:29:22.437548Z","iopub.execute_input":"2023-09-12T10:29:22.43856Z","iopub.status.idle":"2023-09-12T10:29:22.455241Z","shell.execute_reply.started":"2023-09-12T10:29:22.438524Z","shell.execute_reply":"2023-09-12T10:29:22.454035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Parallel(n_jobs = 3)(\n        delayed(process)(patient, size=SIZE, save_folder=SAVE_FOLDER, data_path=TEST_PATH)\n        for patient in tqdm(patients)\n        )","metadata":{"execution":{"iopub.status.busy":"2023-09-12T10:31:44.681365Z","iopub.execute_input":"2023-09-12T10:31:44.682751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random.choice(os.listdir('/kaggle/working/train_images_128_png_complete'))","metadata":{"execution":{"iopub.status.busy":"2023-09-12T10:30:23.181069Z","iopub.status.idle":"2023-09-12T10:30:23.181617Z","shell.execute_reply.started":"2023-09-12T10:30:23.181296Z","shell.execute_reply":"2023-09-12T10:30:23.181315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_file = random.choice(os.listdir(SAVE_FOLDER))\nimg = cv2.imread(SAVE_FOLDER + random_file, 0)\n#cv2.imshow('image', img)\n\nplt.figure(figsize=(15, 15))\nplt.imshow(img, cmap=\"gray\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-12T10:30:23.183243Z","iopub.status.idle":"2023-09-12T10:30:23.184Z","shell.execute_reply.started":"2023-09-12T10:30:23.183794Z","shell.execute_reply":"2023-09-12T10:30:23.183815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar -zcf train_images_128_png_complete.tar.gz /kaggle/working/train_images_128_png_complete/","metadata":{"execution":{"iopub.status.busy":"2023-09-12T10:30:23.185519Z","iopub.status.idle":"2023-09-12T10:30:23.186396Z","shell.execute_reply.started":"2023-09-12T10:30:23.186181Z","shell.execute_reply":"2023-09-12T10:30:23.186203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2. Create a Small Subset of DICOM Images for Prototyping\n\nInstead of working with all DICOM images for all patients in the training dataset, we'll only a small subset of them. We would like to ensure that all injury labels are present in adequate amount in this small dataset.","metadata":{}},{"cell_type":"code","source":"import os\nimport random\nimport shutil\nfrom tqdm.notebook import tqdm\nimport pandas as pd\nrandom.seed(123)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T10:56:44.979746Z","iopub.execute_input":"2023-09-16T10:56:44.980269Z","iopub.status.idle":"2023-09-16T10:56:44.985689Z","shell.execute_reply.started":"2023-09-16T10:56:44.98023Z","shell.execute_reply":"2023-09-16T10:56:44.984734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DICOM_PATH = '/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images/'\nSAVE_PATH = '/kaggle/working/train_images_DICOM_500/'\nnum_patients = 500\nnum_img = 10\nos.makedirs(SAVE_PATH, exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T10:56:44.997394Z","iopub.execute_input":"2023-09-16T10:56:44.997906Z","iopub.status.idle":"2023-09-16T10:56:45.005142Z","shell.execute_reply.started":"2023-09-16T10:56:44.997866Z","shell.execute_reply":"2023-09-16T10:56:45.004263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patients = random.sample(os.listdir(DICOM_PATH), k = num_patients)\n\nfor pat in tqdm(patients):\n    \n    series_id = random.choice(os.listdir(DICOM_PATH + str(pat)))\n    source_folder = os.path.join(DICOM_PATH, pat, series_id)\n    \n    files_to_copy = random.sample(os.listdir(source_folder), k = num_img)\n    \n    for file in files_to_copy:\n        target_filename = os.path.join(SAVE_PATH, str(pat) + '_' + str(series_id) + '_' + file)\n        file_path = os.path.join(source_folder, file)\n        shutil.copy(file_path, target_filename)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-09-16T10:56:45.019936Z","iopub.execute_input":"2023-09-16T10:56:45.021557Z","iopub.status.idle":"2023-09-16T10:56:54.175784Z","shell.execute_reply.started":"2023-09-16T10:56:45.021463Z","shell.execute_reply":"2023-09-16T10:56:54.174028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us look at the distribution of labels in this small DICOM image set.","metadata":{}},{"cell_type":"code","source":"patient_labels = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')\npatients = [int(pat) for pat in patients]\nsubset_patient_labels = patient_labels[patient_labels['patient_id'].isin(patients)]","metadata":{"execution":{"iopub.status.busy":"2023-09-16T10:56:54.17906Z","iopub.execute_input":"2023-09-16T10:56:54.179525Z","iopub.status.idle":"2023-09-16T10:56:54.197089Z","shell.execute_reply.started":"2023-09-16T10:56:54.179488Z","shell.execute_reply":"2023-09-16T10:56:54.194865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_pos_num(df, label):\n    num_entries = df.shape[0]\n    return (df[label] == 1).sum()\n\ninjury_labels = ['bowel_injury', 'extravasation_injury' , 'kidney_low' , 'kidney_high', 'liver_low',\n                 'liver_high', 'spleen_low', 'spleen_high']\n\nprint(f'Total number of patients: {num_patients}')\nfor label in injury_labels:    \n    print(f'Number of patients with {label} = {get_pos_num(subset_patient_labels, label)}')","metadata":{"execution":{"iopub.status.busy":"2023-09-16T10:56:54.199243Z","iopub.execute_input":"2023-09-16T10:56:54.199774Z","iopub.status.idle":"2023-09-16T10:56:54.213038Z","shell.execute_reply.started":"2023-09-16T10:56:54.199728Z","shell.execute_reply":"2023-09-16T10:56:54.21193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!du -sh /kaggle/working/train_images_DICOM_500/","metadata":{"execution":{"iopub.status.busy":"2023-09-16T10:56:54.2165Z","iopub.execute_input":"2023-09-16T10:56:54.216896Z","iopub.status.idle":"2023-09-16T10:56:54.506133Z","shell.execute_reply.started":"2023-09-16T10:56:54.216866Z","shell.execute_reply":"2023-09-16T10:56:54.503903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir(SAVE_PATH))","metadata":{"execution":{"iopub.status.busy":"2023-09-16T10:57:18.680414Z","iopub.execute_input":"2023-09-16T10:57:18.680801Z","iopub.status.idle":"2023-09-16T10:57:18.69498Z","shell.execute_reply.started":"2023-09-16T10:57:18.680774Z","shell.execute_reply":"2023-09-16T10:57:18.693161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. Resize (512x512) PNG to (sizexsize) PNG\n\nWe take the (512x512) PNG datasets provided [here](https://www.kaggle.com/code/theoviel/get-started-quicker-dicom-png-conversion) and convert them to a given size. It's helpful to initally experiment with datasets of small sized images (128x128, 256x256, etc).","metadata":{}},{"cell_type":"code","source":"BASE_DIR = '/kaggle/input/rsna-abdominal-trauma-detection-png-pt2'\nSAVE_DIR = '/kaggle/working/rsna-atd-128-png-pt2/'\nSIZE = 128\nos.makedirs(SAVE_DIR, exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T09:12:11.133631Z","iopub.execute_input":"2023-09-26T09:12:11.13409Z","iopub.status.idle":"2023-09-26T09:12:11.142256Z","shell.execute_reply.started":"2023-09-26T09:12:11.134055Z","shell.execute_reply":"2023-09-26T09:12:11.141044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(fn, size):\n    im = cv2.imread(os.path.join(BASE_DIR, fn))\n    im = cv2.resize(im, (size, size))\n    cv2.imwrite(os.path.join(SAVE_DIR, fn), (im * 255).astype(np.uint8))","metadata":{"execution":{"iopub.status.busy":"2023-09-26T09:12:13.728609Z","iopub.execute_input":"2023-09-26T09:12:13.729081Z","iopub.status.idle":"2023-09-26T09:12:13.736578Z","shell.execute_reply.started":"2023-09-26T09:12:13.729047Z","shell.execute_reply":"2023-09-26T09:12:13.735079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import multiprocessing\n\nmultiprocessing.cpu_count()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T08:26:30.055258Z","iopub.execute_input":"2023-09-26T08:26:30.055636Z","iopub.status.idle":"2023-09-26T08:26:30.062407Z","shell.execute_reply.started":"2023-09-26T08:26:30.055597Z","shell.execute_reply":"2023-09-26T08:26:30.061311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f_list = os.listdir(BASE_DIR)\nParallel(n_jobs = 4)(\n    delayed(process)(fn, SIZE)\n    for fn in tqdm(f_list)\n    )","metadata":{"execution":{"iopub.status.busy":"2023-09-26T09:12:18.601338Z","iopub.execute_input":"2023-09-26T09:12:18.601773Z"},"trusted":true},"execution_count":null,"outputs":[]}]}