{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\nfrom tqdm.notebook import tqdm\n\nfrom PIL import Image","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 1. Preprocessing\n\nConverting DICOM files to png images.\nFrom https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Util Methods\ndef read_xray(path):\n    dicom_file = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    data = apply_voi_lut(dicom_file.pixel_array, dicom_file)\n    #MONOCHROME1 indicates that the greyscale ranges from bright to dark with ascending pixel values, \n    #whereas MONOCHROME2 ranges from dark to bright with ascending pixel values.\n    if dicom_file.PhotometricInterpretation == 'MONOCHROME1':\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data/np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\ndef resize(array, size):\n    im = Image.fromarray(array)\n    #LANCZOS (a high-quality downsampling filter)\n    im = im.resize((size,size),  resample = Image.LANCZOS)\n    return im","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"training_image_ids = []\ndim_0 = []\ndim_1 = []\n\nfor split in ['train','test']:\n    load_dir = f'../input/vinbigdata-chest-xray-abnormalities-detection/{split}/'\n    save_dir = f'/kaggle/tmp/{split}'\n    #Creating save_dirs\n    os.makedirs(save_dir, exist_ok = True)\n    #iterating over each file\n    for file in tqdm(os.listdir(load_dir)):\n        xray = read_xray(load_dir+file)\n        im = resize(xray, size = 512)\n        im.save(save_dir+file.replace('.dicom','.png'))\n        \n        if split == 'train':\n            training_image_ids.append(file.replace('.dicom',''))\n            dim_0.append(xray.shape[0])\n            dim_1.append(xray.shape[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"! tar -zcf train.tar.gz -C \"/kaggle/tmp/train\"\n! tar -zcf test.tar.gz -C \"/kaggle/tmp/test\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.DataFrame({\"image_id\":training_image_ids,\"dim_0\":dim_0,\"dim1\":dim_1})\ndf.to_csv(\"train_metadata.csv\", index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}