{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Introduction\n\nThe goal of this notebook is to preapre the dataset before analysis.\n\n* Resize the dataset\n* Prepare K folds of the dataset."},{"metadata":{},"cell_type":"markdown","source":"# Methods"},{"metadata":{},"cell_type":"markdown","source":"# Initialize"},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport pydicom # To read dicom files","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# See https://www.kaggle.com/bryanb/vinbigdata-chest-x-ray-eda\n\n# Path to dataframe and image folders\nPATH = \"../input/vinbigdata-chest-xray-abnormalities-detection\"\n\n# Import trainset\ntrain = pd.read_csv(os.path.join(PATH, 'train.csv'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# see https://www.kaggle.com/asimzahid/all-you-need-to-know-about-dicom\ntrain.image_id.value_counts().to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# See https://www.kaggle.com/asimzahid/all-you-need-to-know-about-dicom\nimages = train.image_id.nunique()\nprint(f\"There are in total {images} unique images in the train test.\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# See https://pydicom.github.io/pydicom/stable/auto_examples/image_processing/plot_downsize_image.html\nimport pydicom\nfrom pydicom.data import get_testdata_file\n\nprint(__doc__)\n\n# FIXME: add a full-sized MR image in the testing data\nfilename = get_testdata_file('MR_small.dcm')\nds = pydicom.dcmread(filename)\n\n# get the pixel information into a numpy array\ndata = ds.pixel_array\nprint('The image has {} x {} voxels'.format(data.shape[0],\n                                            data.shape[1]))\ndata_downsampling = data[::8, ::8]\nprint('The downsampled image has {} x {} voxels'.format(\n    data_downsampling.shape[0], data_downsampling.shape[1]))\n\n# copy the data back to the original data set\nds.PixelData = data_downsampling.tobytes()\n# update the information regarding the shape of the data array\nds.Rows, ds.Columns = data_downsampling.shape\n\n# print the image information given in the dataset\nprint('The information of the data set after downsampling: \\n')\nprint(ds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# See https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\nimport numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# See https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\nfrom PIL import Image\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# See https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\nfrom tqdm.auto import tqdm\nimage_id = []\ndim0 = []\ndim1 = []\n\nfor split in ['train', 'test']:\n    load_dir = f'../input/vinbigdata-chest-xray-abnormalities-detection/{split}/'\n    save_dir = f'/kaggle/tmp/{split}/'\n\n    os.makedirs(save_dir, exist_ok=True)\n\n    for file in tqdm(os.listdir(load_dir)):\n        # set keep_ratio=True to have original aspect ratio\n        xray = read_xray(load_dir + file)\n        im = resize(xray, size=1024)  \n        im.save(save_dir + file.replace('dicom', 'png'))\n        \n        if split == 'train':\n            image_id.append(file.replace('.dicom', ''))\n            dim0.append(xray.shape[0])\n            dim1.append(xray.shape[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n!tar -zcf train.tar.gz -C \"/kaggle/tmp/train/\" .\n!tar -zcf test.tar.gz -C \"/kaggle/tmp/test/\" .","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.DataFrame.from_dict({'image_id': image_id, 'dim0': dim0, 'dim1': dim1})\ndf.to_csv('train_meta.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# References\n\n## Discussions\n\n* https://www.kaggle.com/c/osic-pulmonary-fibrosis-progression/discussion/174434\n\n## Notebooks\n\n* https://www.kaggle.com/asimzahid/all-you-need-to-know-about-dicom\n* https://www.kaggle.com/bhallaakshit/dicom-wrangling-and-enhancement\n* https://www.kaggle.com/notvuvko/vinbigdata-process-and-resize-original-ratio\n* https://www.kaggle.com/minmin102/vinbigdata-chest-x-ray-ad-preprocess-dicom\n* https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-png-256x256\n* https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n* https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n\n## stackoverflow\n\n* https://stackoverflow.com/questions/55560243/resize-a-dicom-image-in-python\n\n## Websites\n\n* https://pydicom.github.io/pydicom/stable/auto_examples/image_processing/plot_downsize_image.html\n\n## Books\n\n* Approaching (almost) any machine learning problem by abhishek thakur\n\n## Videos"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}