{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Chest Ray Segmentation","metadata":{}},{"cell_type":"markdown","source":"> ","metadata":{}},{"cell_type":"markdown","source":"![](https://i.imgur.com/QWmbhXx.png)","metadata":{}},{"cell_type":"markdown","source":"# 1. Dicom to Numpy array","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport pydicom\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nfrom skimage import exposure\nimport cv2\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-09T12:02:41.029794Z","iopub.execute_input":"2023-07-09T12:02:41.030389Z","iopub.status.idle":"2023-07-09T12:02:41.875395Z","shell.execute_reply.started":"2023-07-09T12:02:41.03034Z","shell.execute_reply":"2023-07-09T12:02:41.874343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All images in dataset are DICOM format. So we need to convert data from DICOM to numpy array. Original dicom2array function in [raddar's notebook](https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way)","metadata":{}},{"cell_type":"code","source":"dataset_dir = '../input/vinbigdata-chest-xray-abnormalities-detection'","metadata":{"execution":{"iopub.status.busy":"2023-07-09T12:02:41.877757Z","iopub.execute_input":"2023-07-09T12:02:41.878322Z","iopub.status.idle":"2023-07-09T12:02:41.88369Z","shell.execute_reply.started":"2023-07-09T12:02:41.878272Z","shell.execute_reply":"2023-07-09T12:02:41.882695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef dicom2array(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to\n    # transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n        \n    \ndef plot_img(img, size=(7, 7), is_rgb=True, title=\"\", cmap='gray'):\n    plt.figure(figsize=size)\n    plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n\n\ndef plot_imgs(imgs, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500,500)):\n    rows = len(imgs)//cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows, cols, i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n    \n# def draw_bboxes(img, boxes, thickness=10, color=(255, 0, 0), img_size=(500,500)):\n#     img_copy = img.copy()\n#     if len(img_copy.shape) == 2:\n#         img_copy = np.stack([img_copy, img_copy, img_copy], axis=-1)\n#     for box in boxes:\n#         img_copy = cv2.rectangle(\n#             img_copy,\n#             (int(box[0]), int(box[1])),\n#             (int(box[2]), int(box[3])),\n#             color, thickness)\n#     if img_size is not None:\n#         img_copy = cv2.resize(img_copy, img_size)\n#     return img_copy","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2023-07-09T12:02:41.885149Z","iopub.execute_input":"2023-07-09T12:02:41.8855Z","iopub.status.idle":"2023-07-09T12:02:41.905251Z","shell.execute_reply.started":"2023-07-09T12:02:41.885465Z","shell.execute_reply":"2023-07-09T12:02:41.904056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_paths = glob(f'{dataset_dir}/train/*.dicom')\nimgs = [dicom2array(path) for path in dicom_paths[:4]]\nplot_imgs(imgs)","metadata":{"execution":{"iopub.status.busy":"2023-07-09T12:02:41.906993Z","iopub.execute_input":"2023-07-09T12:02:41.907667Z","iopub.status.idle":"2023-07-09T12:02:52.303342Z","shell.execute_reply.started":"2023-07-09T12:02:41.907616Z","shell.execute_reply":"2023-07-09T12:02:52.302012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Maybe, you can try some preprocess like equalize histogram. You can see the difference between before and after","metadata":{}},{"cell_type":"code","source":"\nimgs = [exposure.equalize_hist(img) for img in imgs]\nplot_imgs(imgs)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T12:02:52.305818Z","iopub.execute_input":"2023-07-09T12:02:52.306407Z","iopub.status.idle":"2023-07-09T12:02:54.490993Z","shell.execute_reply.started":"2023-07-09T12:02:52.30635Z","shell.execute_reply":"2023-07-09T12:02:54.490077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. EDA csv","metadata":{}},{"cell_type":"markdown","source":"### Now, we will try some EDA steps to find out important features in this data set","metadata":{}},{"cell_type":"code","source":"from bokeh.plotting import figure as bokeh_figure\nfrom bokeh.io import output_notebook, show, output_file\nfrom bokeh.models import ColumnDataSource, HoverTool, Panel\nfrom bokeh.models.widgets import Tabs\nimport pandas as pd\nfrom PIL import Image\nfrom sklearn import preprocessing\nimport random\nfrom random import randint\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-09T12:02:54.492528Z","iopub.execute_input":"2023-07-09T12:02:54.493115Z","iopub.status.idle":"2023-07-09T12:02:55.769907Z","shell.execute_reply.started":"2023-07-09T12:02:54.493075Z","shell.execute_reply":"2023-07-09T12:02:55.768819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_bbox_area(row):\n    return (row['x_max']-row['x_min'])*(row['y_max']-row['y_min'])\n\ntrain_df = pd.read_csv(f'{dataset_dir}/train.csv')\nle = preprocessing.LabelEncoder()  # encode rad_id\ntrain_df['rad_label'] = le.fit_transform(train_df['rad_id'])\n\nfinding_df = train_df[train_df['class_name'] != 'No finding']\nfinding_df['bbox_area'] = finding_df.apply(get_bbox_area, axis=1)\nfinding_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-09T12:02:55.771739Z","iopub.execute_input":"2023-07-09T12:02:55.772274Z","iopub.status.idle":"2023-07-09T12:02:57.424337Z","shell.execute_reply.started":"2023-07-09T12:02:55.772235Z","shell.execute_reply":"2023-07-09T12:02:57.423403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1 Plot bounding box","metadata":{}},{"cell_type":"code","source":"imgs = []\nimg_ids = finding_df['image_id'].values\nclass_ids = finding_df['class_id'].unique()\n\n# map label_id to specify color\nlabel2color = {class_id:[randint(0,255) for i in range(3)] for class_id in class_ids}\nthickness = 3\nscale = 5\n\n\nfor i in range(8):\n    img_id = random.choice(img_ids)\n    img_path = f'{dataset_dir}/train/{img_id}.dicom'\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    boxes = finding_df.loc[finding_df['image_id'] == img_id, ['x_min', 'y_min', 'x_max', 'y_max']].values/scale\n    labels = finding_df.loc[finding_df['image_id'] == img_id, ['class_id']].values.squeeze()\n    \n    for label_id, box in zip(labels, boxes):\n        color = label2color[label_id]\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n    )\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)","metadata":{"execution":{"iopub.status.busy":"2023-07-09T12:02:57.42592Z","iopub.execute_input":"2023-07-09T12:02:57.426312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can see that: in each image, there are many overlapping boxes. Note that a key part of this competition is working with ground truth from multiple radiologists. I guess it is a key to get best rank in this competition if you handle it skillfully.","metadata":{}},{"cell_type":"markdown","source":"## 2.2 Plot histogram","metadata":{}},{"cell_type":"markdown","source":"We will try to plot some histograms.  The hist_hover function allows you to interact very well with chart","metadata":{}},{"cell_type":"code","source":"def hist_hover(dataframe, column, color=[\"#94c8d8\", \"#ea5e51\"], bins=30, title=\"\", value_range=None):\n    \"\"\"\n    Plot histogram\n    \"\"\"\n    hist, edges = np.histogram(dataframe[column], bins=bins, range=value_range)\n    hist_frame = pd.DataFrame({\n        column: hist,\n        \"left\": edges[:-1],\n        \"right\": edges[1:]\n    })\n    hist_frame[\"interval\"] = [\"%d to %d\" %\n                              (left, right) for left, right in zip(edges[:-1], edges[1:])]\n    src = ColumnDataSource(hist_frame)\n    plot = bokeh_figure(\n        plot_height=400, plot_width=600,\n        title=title, x_axis_label=column,\n        y_axis_label=\"Count\"\n    )\n    plot.quad(\n        bottom=0, top=column, left=\"left\", right=\"right\",\n        source=src, fill_color=color[0], line_color=\"#35838d\",\n        fill_alpha=0.7, hover_fill_alpha=0.7,\n        hover_fill_color=color[1]\n    )\n    hover = HoverTool(\n        tooltips=[(\"Interval\", \"@interval\"), (\"Count\", str(f\"@{column}\"))]\n    )\n    plot.add_tools(hover)\n    output_notebook()\n    show(plot)\n    \n    \nhist_hover(train_df, column='class_id')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"you can see the imbalance between image qualtity of each class","metadata":{}},{"cell_type":"code","source":"#Note that a key part of this competition is working with ground truth from multiple radiologists.\nhist_hover(train_df, column='rad_label')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The imbalance between image qualtity of each radiologist","metadata":{}},{"cell_type":"code","source":"## histogram of bbox area\nhist_hover(finding_df, column='bbox_area')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After some EDA steps, we recognize that the dataset fairly imbalance in many aspects. Maybe, we need to use some augmentation method to resolve the problem.\n","metadata":{}},{"cell_type":"markdown","source":"# 3. Don't forget to upvote :D","metadata":{}}]}