{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook I resized the dataset from 400GB to ~7.5GB doing this step:\n\n- Resized the DCM images from 512x512 to 256x256\n- Reduced the slice tickness to only 5 millimeters, skipping the intermediate scans (slice tickness range from 0.5 to 5mm)\n- Encoded and standardized the DCM images to JPEG\n\n### You can download directly the reduced dataset from here\n\nhttps://www.kaggle.com/datasets/alenic/rsna-2023-atd-reduced-256-5mm\n","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nimport cv2\nimport pydicom\nfrom joblib import Parallel, delayed\n\nJ = os.path.join\noriginal_dataset_root = \"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_images\"\nout_dataset_root = os.path.join(\"rsna-abdominal-2023\")\n\nN_JOBS = 2\n\n# Perform only the first iteration\nBREAK = True #  <- #!!Set to False if you want to process the entire dataset)!!\n\n# Resize images\nRESIZE = True\nSIZE = 256\n# Reduce Slice Tickness to 5 millimeters\nTICK = 5   # mm","metadata":{"execution":{"iopub.status.busy":"2023-08-19T22:36:59.586001Z","iopub.execute_input":"2023-08-19T22:36:59.586454Z","iopub.status.idle":"2023-08-19T22:36:59.592957Z","shell.execute_reply.started":"2023-08-19T22:36:59.586415Z","shell.execute_reply":"2023-08-19T22:36:59.59155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get DICOM meta-data\ndf_dicom = pd.read_parquet(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/train_dicom_tags.parquet\")\n# Gest series folder\ndf_dicom[\"PatientID\"] = df_dicom[\"PatientID\"].astype(str)\ndf_dicom[\"serie\"] = df_dicom[\"SeriesInstanceUID\"].apply(lambda x: x.split(\".\")[-1])\ndf_dicom = df_dicom.set_index([\"PatientID\", \"serie\"]).sort_index()","metadata":{"execution":{"iopub.status.busy":"2023-08-19T22:29:54.080044Z","iopub.execute_input":"2023-08-19T22:29:54.080461Z","iopub.status.idle":"2023-08-19T22:30:05.711606Z","shell.execute_reply.started":"2023-08-19T22:29:54.080425Z","shell.execute_reply":"2023-08-19T22:30:05.710288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the values of Slice Tickness in millimeters from DICOM meta-data\ndf_dicom.SliceThickness.value_counts().sort_index()","metadata":{"execution":{"iopub.status.busy":"2023-08-19T22:31:14.629129Z","iopub.execute_input":"2023-08-19T22:31:14.62959Z","iopub.status.idle":"2023-08-19T22:31:14.668445Z","shell.execute_reply.started":"2023-08-19T22:31:14.629554Z","shell.execute_reply":"2023-08-19T22:31:14.667304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Dataset transformation","metadata":{}},{"cell_type":"markdown","source":"The dicom_to_image code taken from this noteobok (by @harsha1999): https://www.kaggle.com/code/harsha1999/eda-with-3d-visualization","metadata":{}},{"cell_type":"code","source":"def dicom_to_image(dicom_image):\n    \"\"\"\n    Read the dicom file and preprocess appropriately.\n    \"\"\"\n    pixel_array = dicom_image.pixel_array\n    \n    if dicom_image.PixelRepresentation == 1:\n        bit_shift = dicom_image.BitsAllocated - dicom_image.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dicom_image)\n    \n    if dicom_image.PhotometricInterpretation == \"MONOCHROME1\":\n        pixel_array = 1 - pixel_array\n    \n    # transform to hounsfield units\n    intercept = dicom_image.RescaleIntercept\n    slope = dicom_image.RescaleSlope\n    pixel_array = pixel_array * slope + intercept\n    \n    # windowing\n    window_center = int(dicom_image.WindowCenter)\n    window_width = int(dicom_image.WindowWidth)\n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    pixel_array = pixel_array.copy()\n    pixel_array[pixel_array < img_min] = img_min\n    pixel_array[pixel_array > img_max] = img_max\n    \n    # normalization\n    pixel_array = (pixel_array - pixel_array.min())/(pixel_array.max() - pixel_array.min())\n    \n    return (pixel_array * 255).astype(np.uint8)","metadata":{"execution":{"iopub.status.busy":"2023-08-19T22:32:43.489007Z","iopub.execute_input":"2023-08-19T22:32:43.489431Z","iopub.status.idle":"2023-08-19T22:32:43.499549Z","shell.execute_reply.started":"2023-08-19T22:32:43.489395Z","shell.execute_reply":"2023-08-19T22:32:43.498236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.1 Preprocess","metadata":{}},{"cell_type":"code","source":"\nif not RESIZE: SIZE=512\n\nout_folder = os.path.join(out_dataset_root, f\"reduced_{SIZE}_tickness_{TICK}\")\n\ndef process(dcm_path, p, s):\n    dicom_image = pydicom.read_file(dcm_path)\n    dcm_file = os.path.basename(dcm_path)\n    # read DICOM\n    img = dicom_to_image(dicom_image)\n\n    # Resize\n    if RESIZE:\n        img = cv2.resize(img, (SIZE, SIZE))\n\n    out_path = J(out_folder, p, s, dcm_file.replace(\".dcm\",\".jpeg\"))\n    cv2.imwrite(out_path, img)\n\n\n\np_ids = os.listdir(original_dataset_root)\n\nfor p in tqdm(p_ids):\n    p_folder = J(original_dataset_root, p)\n    s_ids = os.listdir(p_folder)\n    for s in s_ids:\n        s_folder = J(p_folder, s)\n        os.makedirs(J(out_folder, p, s), exist_ok=True)\n\n        # sort dcm files\n        dcms = os.listdir(s_folder)\n\n        dcms_d = {int(x.replace(\".dcm\",\"\")): x for x in dcms}\n        dcms = sorted(dcms_d)\n        dcms = [str(x)+\".dcm\" for x in dcms]\n        \n        n_d = len(dcms)\n        \n        curr_tick = df_dicom.loc[(p, s), \"SliceThickness\"].mean()\n        step = round(TICK/curr_tick)\n\n        dcm_paths = []\n        for i in range(0, n_d, step):\n            dcm_paths.append(J(s_folder, dcms[i]))\n        \n        _ = Parallel(n_jobs=N_JOBS)(\n            delayed(process)(path, p, s) for path in dcm_paths\n        )\n\n\n    if BREAK: break\n","metadata":{"execution":{"iopub.status.busy":"2023-08-19T22:38:01.528988Z","iopub.execute_input":"2023-08-19T22:38:01.530176Z","iopub.status.idle":"2023-08-19T22:38:06.009756Z","shell.execute_reply.started":"2023-08-19T22:38:01.530139Z","shell.execute_reply":"2023-08-19T22:38:06.008711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}