{"cells":[{"metadata":{},"cell_type":"markdown","source":"# EDA Chest X-ray Abnormalities Detection"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"from IPython.display import HTML\nHTML('<center><iframe width=\"650\" height=\"450\" src=\"https://www.youtube.com/embed/PRS_CXprri0\" frameborder=\"0\" allow=\"accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture\" allowfullscreen></iframe></center>')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Import Libraries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\nimport gc\nimport numpy as np # linear algebra\nimport pandas # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n#import matplotlib.patches as ptc\nimport plotly.graph_objects as go\nimport seaborn as sns\n%matplotlib inline\nfrom pydicom import dcmread\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n'''\nfrom functools import partial\nimport multiprocessing as mpc\nfrom joblib import Parallel, delayed\n'''\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"PATH = \"../input/vinbigdata-chest-xray-abnormalities-detection/\"\ntrain_df = pandas.read_csv(os.path.join(PATH, 'train.csv'))\ntrain_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"Rows, Cols = train_df.shape\nprint(f'There are {Rows} Rows and {Cols} columns in train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.class_id.nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.groupby(['class_name', 'class_id']).agg({'count'})['image_id'].sort_values(by='count').rename(columns={0:\"Unique Values\"}).style.background_gradient(cmap=\"plasma\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10, 10))\nsns.pairplot(train_df, hue='class_name')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.image_id.value_counts().to_frame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"images = train_df.image_id.nunique()\nprint(f\"There are in total {images} unique images in the train test.\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Read Dicom image"},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"di = dcmread('../input/vinbigdata-chest-xray-abnormalities-detection/train/000434271f63a053c4128a0ba6352c7f.dicom')\ndi","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"View Dicom image"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16,6))\nx = plt.imshow(di.pixel_array, 'gray')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16,6))\nx = plt.imshow(di.pixel_array, cmap=plt.cm.bone)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"plt.figure(figsize=(16,6))\nx = plt.imshow(di.pixel_array, cmap=plt.cm.gist_ncar)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Class Distribution"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(28, 8))\nsns.countplot(x=\"class_name\", orient=\"h\", data=train_df)\nplt.title(\"Class Distribution\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"def plot_distribution_classes(x_values, y_values, title):\n    \n    #colors = ['rgb(26, 118, 255)',] * 15\n    #colors[0] = 'lightslategray'\n\n    fig = go.Figure(data=[go.Bar(\n        x=x_values, \n        y=y_values,\n        text=y_values\n        #marker_color=colors\n    )])\n\n    fig.update_layout(height=400, width=700, title_text=title)\n    fig.update_xaxes(type=\"category\")\n\n    fig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"train_df.class_name.value_counts().to_frame().rename(columns={0:\"Unique Values\"}).style.background_gradient(cmap=\"plasma\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"indexes = train_df.class_name.unique()\ncounts = train_df.class_name.value_counts()\n\nsorted_dict = dict(zip(indexes, counts))\nsorted_dict = {k: v for k, v in sorted(sorted_dict.items(), key=lambda item: item[1], reverse = True)}\n\nx = list(sorted_dict.keys())\ny = list(sorted_dict.values())\n\nplot_distribution_classes(x, y, \n                          title=\"Distribution of radiographic observations\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"As we can see there is a class imbalance problem. We need to augment the data to address this problem."},{"metadata":{},"cell_type":"markdown","source":"# Radiologiet Distribution"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"train_df.rad_id.value_counts().to_frame().rename(columns={0:\"Unique Values\"}).style.background_gradient(cmap=\"plasma\")","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"indexes = train_df.rad_id.unique()\ncounts = train_df.rad_id.value_counts()\n\nsorted_dict = dict(zip(indexes, counts))\nsorted_dict = {k: v for k, v in sorted(sorted_dict.items(), key=lambda item: item[1], reverse = True)}\n\nx = list(sorted_dict.keys())\ny = list(sorted_dict.values())\n\nplot_distribution_classes(x, y, \n                          title=\"Distribution of Annotations by Radioloiest\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# FastAI to process DICOMs "},{"metadata":{},"cell_type":"markdown","source":"### DICOM metadata to Dataframe --> Pikle"},{"metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"cell_type":"code","source":"!pip install -Uqq fastai","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"cell_type":"code","source":"from fastai.basics import *\nfrom fastai.callback.all import *\nfrom fastai.vision.all import *\nfrom fastai.medical.imaging import *\n\nimport pydicom,kornia,skimage\nfrom pydicom.dataset import Dataset as DcmDataset\nfrom pydicom.tag import BaseTag as DcmTag\nfrom pydicom.multival import MultiValue as DcmMultiValue\nfrom PIL import Image\n\ntry:\n    import cv2\n    cv2.setNumThreads(0)\nexcept: pass","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path = Path('../input/vinbigdata-chest-xray-abnormalities-detection')\ntrain_imgs = path/'train'\ntrain_dicom = get_dicom_files(train_imgs)\ndicom_dataframe = pd.DataFrame.from_dicoms(train_dicom, window=dicom_windows.lungs, px_summ=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Write DICOM metadata into pkl for fast processing"},{"metadata":{"trusted":true},"cell_type":"code","source":"dicom_dataframe.to_pickle('./dicom_dataframe_pickle.pkl')\ndicom_dataframe.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Read metadata from pickle"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"dicom_dataframe = pd.read_pickle('./dicom_dataframe_pickle.pkl')\ndicom_dataframe.shape # should be 15k by 29","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"View DICOM metadata into dataframe"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"dicom_dataframe","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Save DICOM to Image"},{"metadata":{},"cell_type":"markdown","source":"## DICOM to PNG"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"import pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom PIL import Image\nfrom tqdm.auto import tqdm\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"def resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"image_id = []\ndim0 = []\ndim1 = []\n\nfor split in ['train', 'test']:\n    load_dir = f'../input/vinbigdata-chest-xray-abnormalities-detection/{split}/'\n    save_dir = f'/kaggle/tmp/{split}/'\n\n    os.makedirs(save_dir, exist_ok=True)\n\n    for file in tqdm(os.listdir(load_dir)):\n        # set keep_ratio=True to have original aspect ratio\n        xray = read_xray(load_dir + file)\n        im = resize(xray, size=1024)  \n        im.save(save_dir + file.replace('dicom', 'png'))\n        \n        if split == 'train':\n            image_id.append(file.replace('.dicom', ''))\n            dim0.append(xray.shape[0])\n            dim1.append(xray.shape[1])","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"%%time\n!tar -zcf train.tar.gz -C \"/kaggle/tmp/train/\" .\n!tar -zcf test.tar.gz -C \"/kaggle/tmp/test/\"","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"df = pd.DataFrame.from_dict({'image_id': image_id, 'dim0': dim0, 'dim1': dim1})\ndf.to_csv('train_meta.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"References: \n1. https://docs.fast.ai/medical.imaging\n2. https://www.kaggle.com/crained/vinbigdata-fastai-get-started\n3. https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n4. https://www.kaggle.com/c/vinbigdata-chest-xray-abnormalities-detection/discussion/207955\n5. https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-png-1024x1024\n\nThanks a million to the community.\n\nUpcoming:\n\nworking on fastai build-in save_jpg() method to convert dicom to jpg. \n\nAny advice and contriutions would be appreciated.\nThank you.\n\n\nThis notebook is a starter guide for ones who want to start with DICOM.\n\n# Work In progress"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}