{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg # For better performance in turning dicom to png files","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport pydicom\nfrom matplotlib import pyplot as plt\nfrom tqdm.notebook import tqdm\nimport cv2\nimport glob\nfrom joblib import Parallel, delayed\nimport os,time\nimport gdcm","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def standardize_pixel_array(dcm: pydicom.dataset.FileDataset) -> np.ndarray:\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    pixel_array = dcm.pixel_array\n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n    return pixel_array","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Before run this code add rsna-breast-cancer dataset\ndata_set_address= '/kaggle/input/rsna-2023-abdominal-trauma-detection/'\ntrain_images = sorted(glob.glob(f'{data_set_address}train_images/*/*/*.dcm'))\n# Use glob to work easier on files in directories\nnumber_of_train_images = len(train_images)\nprint('Number of train_images is :',number_of_train_images) # xepceted:54706\n\n# define a fucntion to read dicom file and return it's pixel\ndef read_dicom(dicom_file):\n    dicom = pydicom.dcmread(dicom_file) # read dciom files\n    dicom_pixel = dicom.pixel_array # read the pixel of images\n\n    dicom_pixel = (dicom_pixel - dicom_pixel.min()) / (dicom_pixel.max() - dicom_pixel.min()) # Standardie with transferig to [0,1] space\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        dicom_pixel = 1 - dicom_pixel # a special kind of format in dicom files, you can search it on the net\n    \n   \n    return dicom_pixel","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop_dicom(dicom_file):\n    \n    result= cv2.connectedComponentsWithStats((dicom_file > 0.005).astype(np.uint8)[:, :], 8, cv2.CV_32S)\n    # result has 4 diffrenet outputs. You can see different outputs here.\n    # But for reconizing the black color(0) from others(1), we just need result[2] , called stat matrix\n    stat_matrix = result[2] # include left, top, width, height, area_size columns\n    # the second row shows the boxes of pixels which are different from background.Also, we don't need are_size cloumn. So:\n    second_row = stat_matrix[1:, 4].argmax() + 1\n    x1, y1, w, h = stat_matrix[second_row][:4]\n    x2 = x1 + w\n    y2 = y1 + h\n    cropped_dicom = dicom_file[y1: y2, x1: x2]\n    return cropped_dicom","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = train_images[:2]\nfor file in files:\n    patient_id = file.split('/')[-3] # split the patient_id from the whole address\n    image_id = file.split('/')[-1][:-4] # split the image_id from the whole address\n    \n\n    # Plot Orginial iamge\n    plt.figure(figsize=(8,8))\n    dicom_pixel = read_dicom(dicom_file=file)\n    plt.imshow(dicom_pixel,cmap='gray')\n    plt.title(f\"Original image:{patient_id}_{image_id}\",fontsize=16)\n    plt.show()\n    \n    # Plot cropped image\n    plt.figure(figsize=(8,8))\n    cropped_iamge = crop_dicom(dicom_file=dicom_pixel)\n    plt.imshow(cropped_iamge,cmap='gray')\n    plt.title(f\"Cropped image:{patient_id}_{image_id}\",fontsize=16)\n    plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_FOLDER = \"/kaggle/working/images/\"\nos.makedirs(SAVE_FOLDER, exist_ok=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dcm_to_png(file,size=384,save_folder=\"/kaggle/working/\", extension ='jpg'):\n    patient_id = file.split('/')[-3]\n    series = file.split('/')[-2]\n    image_id = file.split('/')[-1][:-4]\n    dicom_pixel = read_dicom(dicom_file=file) # read dicom image pixels\n#     cropped_dicom = crop_dicom(dicom_file=dicom_pixel) # crop the dicom image\n    resized_img = cv2.resize(dicom_pixel,dsize=(size,size)) # resize it to specific size\n    os.makedirs(save_folder +\"images/\"+f\"{patient_id}/\", exist_ok=True)\n    os.makedirs(save_folder +\"images/\"+f\"{patient_id}/{series}/\", exist_ok=True)\n    cv2.imwrite(save_folder +\"images/\" + f\"{patient_id}/{series}/{image_id}.{extension}\", (resized_img*255).astype(np.uint8))\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use parallel computing to spend less time on turning dcm to png file\n# if you want to test just some files, please use train_images[:10] instead of train_images, including the whole data\nstart = time.time()\n_ = Parallel(n_jobs=4)(\n    delayed(dcm_to_png)(file, size=384)\n    for file in tqdm(train_images)  # you can use train_images[:100] for testing\n)\nstop = time.time()\nprint('This process took you:',stop-start)\nprint('Now, you can have access to cropped files wtih png format')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\n    \ndef zipdir(path, ziph):\n    # ziph is zipfile handle\n    for root, dirs, files in os.walk(path):\n        for file in files:\n            ziph.write(os.path.join(root, file), \n                       os.path.relpath(os.path.join(root, file), \n                                       os.path.join(path, '..')))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with zipfile.ZipFile('train.zip', 'w', zipfile.ZIP_DEFLATED) as zipf:\n    zipdir(SAVE_FOLDER, zipf)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf SAVE_FOLDER","metadata":{},"execution_count":null,"outputs":[]}]}