{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nimport pydicom\nimport glob as glob\nfrom skimage import exposure\n%matplotlib inline\nimport seaborn as sns\nimport matplotlib\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Understanding the Dataset\n\n**The dataset comprises 18,000 postero-anterior (PA) CXR scans in DICOM format, which were de-identified to protect patient privacy. All images were labeled by a panel of experienced radiologists for the presence of 14 critical radiographic findings as listed below:**\n\n1. 0 - Aortic enlargement\n2. 1 - Atelectasis\n3. 2 - Calcification\n4. 3 - Cardiomegaly\n5. 4 - Consolidation\n6. 5 - ILD\n7. 6 - Infiltration\n8. 7 - Lung Opacity\n9. 8 - Nodule/Mass\n10. 9 - Other lesion\n11. 10 - Pleural effusion\n12. 11 - Pleural thickening\n13. 12 - Pneumothorax\n14. 13 - Pulmonary fibrosis\n\n**The \"No finding\" observation (14) was intended to capture the absence of all findings above.**"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/vinbigdata-chest-xray-abnormalities-detection/train.csv')\ntrain_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head(6)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Let's check the Class_name labels"},{"metadata":{"trusted":true},"cell_type":"code","source":"counts = train_df['class_name'].value_counts()\nplt.figure(figsize=(15,5))\ncounts.plot(kind='barh')\nplt.tight_layout()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Image Datasets "},{"metadata":{"trusted":true},"cell_type":"code","source":"dataset_dir = '../input/vinbigdata-chest-xray-abnormalities-detection/'\ntrain_imgs = '../input/vinbigdata-chest-xray-abnormalities-detection/train/'\ntest_imgs = '../input/vinbigdata-chest-xray-abnormalities-detection/test/'\n\nprint(\"Training samples : {} \".format(len(os.listdir(train_imgs))))\nprint(\"Test samples : {} \".format(len(os.listdir(test_imgs))))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**I will be implementing some processes used in [Trung Thann Ngyuyen](https://www.kaggle.com/trungthanhnguyen0502/eda-vinbigdata-chest-x-ray-abnormalities/)'s notebook**"},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom PIL import Image\n\nlbl = LabelEncoder()\ntrain_df['rad_label'] = lbl.fit_transform(train_df['rad_id'])\ntrain_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.isna().sum().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Defining Bounding Box Area"},{"metadata":{"trusted":true},"cell_type":"code","source":"def bbox_area(row):\n    return (row['x_max']-row['x_min'])*(row['y_max']-row['y_min'])\nfinding_df = train_df[train_df['class_name']!='No finding']\nfinding_df['bbox_area'] = finding_df.apply(bbox_area, axis=1)\nfinding_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def dicom_to_array(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.read_file(path)\n    \n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    if fix_monochrome and dicom.PhotometricInterpretation == 'MONOCHROME1':\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data*255).astype(np.uint8)\n    return data\n\ndef plot_imgs(imgs, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500,500)):\n    rows = len(imgs)//cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows, cols, i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Exploring arrays"},{"metadata":{"trusted":true},"cell_type":"code","source":"img = dicom_to_array('../input/vinbigdata-chest-xray-abnormalities-detection/train/00053190460d56c53cc3e57321387478.dicom')\nimg","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Plotting Bounding Boxes"},{"metadata":{"trusted":true},"cell_type":"code","source":"import random\nfrom random import randint\n\nimgs = []\nimg_ids = finding_df['image_id'].values\nclass_ids = finding_df['class_id'].unique()\n\n# map label_id to specify color\nlabel2color = {class_id:[randint(0,255) for i in range(3)] for class_id in class_ids}\nthickness = 3\nscale = 5\n\n\nfor i in range(8):\n    img_id = random.choice(img_ids)\n    img_path = f'{dataset_dir}/train/{img_id}.dicom'\n    img = dicom_to_array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    boxes = finding_df.loc[finding_df['image_id'] == img_id, ['x_min', 'y_min', 'x_max', 'y_max']].values/scale\n    labels = finding_df.loc[finding_df['image_id'] == img_id, ['class_id']].values.squeeze()\n    \n    for label_id, box in zip(labels, boxes):\n        color = label2color[label_id]\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n    )\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)\nplt.tight_layout()\nplt.axis('off')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.pairplot(train_df, hue='class_name')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Visualizing the Images through CLAHE Normalization\n\n**This method produces sharper images and is quite often used in chest X-ray research. This generates view, which radiologist would not see in his standard workplace. However, it closely resembles the \"bone-enhanced\" view in some X-rays done (usually due to broken ribs).**"},{"metadata":{"trusted":true},"cell_type":"code","source":"dicom_paths = glob.glob(f'{dataset_dir}/train/*.dicom')\nimgs = [dicom_to_array(path) for path in dicom_paths[:4]]\nplot_imgs(imgs)\n\n\n## Maybe, you can try some preprocess like equalize histogram.\n## You can see the difference between before and after\nimgs = [exposure.equalize_adapthist(img) for img in imgs]\nplot_imgs(imgs)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Bounding Boxes with Diseases"},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_example(idx_list):\n    fig, axs = plt.subplots(1, 3, figsize=(15, 10))\n    fig.subplots_adjust(hspace = .1, wspace=.1)\n    axs = axs.ravel()\n    for i in range(3):\n        image_id = train_df.loc[idx_list[i], 'image_id']\n        data_file = pydicom.dcmread(dataset_dir+'train/'+image_id+'.dicom')\n        img = data_file.pixel_array\n        axs[i].imshow(img, cmap=plt.cm.bone)\n        axs[i].set_title(train_df.loc[idx_list[i], 'class_name'])\n        axs[i].set_xticklabels([])\n        axs[i].set_yticklabels([])\n        if train_df.loc[idx_list[i], 'class_name'] != 'No finding':\n            bbox = [train_df.loc[idx_list[i], 'x_min'],\n                    train_df.loc[idx_list[i], 'y_min'],\n                    train_df.loc[idx_list[i], 'x_max'],\n                    train_df.loc[idx_list[i], 'y_max']]\n            p = matplotlib.patches.Rectangle((bbox[0], bbox[1]),\n                                             bbox[2]-bbox[0],\n                                             bbox[3]-bbox[1],\n                                             ec='r', fc='none', lw=2.)\n            axs[i].add_patch(p)\n            \nfor num in range(15):\n    idx_list = train_df[train_df['class_id']==num][0:3].index.values\n    plot_example(idx_list)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# GroupKFold"},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import GroupKFold, train_test_split\n\ntrain_df = train_df[train_df['class_id'] != 14].reset_index(drop=True)\n\ngkf  = GroupKFold(n_splits = 5)\ntrain_df['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(train_df, groups = train_df.image_id.tolist())):\n    train_df.loc[val_idx, 'fold'] = fold\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train, test = train_test_split(train_df, test_size = 0.2, random_state = 45)\nprint(train.shape)\nprint(test.shape)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Submission File"},{"metadata":{"trusted":true},"cell_type":"code","source":"subs_df = pd.read_csv('../input/vinbigdata-chest-xray-abnormalities-detection/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# WORK IN PROGRESS"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}