{"cells":[{"metadata":{},"cell_type":"markdown","source":"# TODO: explain the use of different of windows in Xray"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"%matplotlib inline\nimport os\nimport sys\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sys.path.insert(0, '../input/kaggledicom')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from collections import namedtuple  \nfrom src.utils import misc\nfrom src.preprocess.dicom_to_dataframe import create_record","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pydicom\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torch\nimport torchvision\nimport torchvision.transforms as transforms","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DATA_ROOT = \"../input/vinbigdata-chest-xray-abnormalities-detection/\"\nTRAIN_DIR = os.path.join(DATA_ROOT, \"train\")\nTEST_DIR = os.path.join(DATA_ROOT, \"test\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"colorpal = ['red', 'green', 'blue']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Utils function"},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_groupby_info(df):\n    image_group = []\n    for image_id, pd_frame in df.groupby('image_id'):\n        image_group.append(image_id)\n\n    imageAncClass_group = []\n    for pair, pd_frame in df.groupby(['image_id', 'class_name']):\n        imageAncClass_group.append(pair)\n    print(f\"Considering the data with no finding\")\n    print(f\"length of number of image_id vs length of image_id and class_name = {len(image_group), len(imageAncClass_group)}\")\n    return len(imageAncClass_group)/len(image_group)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Train dataset\n## Ratio of different class\n## total number of image in train dataset = 15000\n## total number of image in public test dataset = 3000\n\n## Number of class = 15"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images = os.listdir(TRAIN_DIR)\nprint(f\"number of image in public test = {len(train_images)}\")\n\ntest_images = os.listdir(TEST_DIR)\nprint(f\"number of image in public test = {len(test_images)}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(os.path.join(DATA_ROOT,'train.csv')).fillna(-1)\n\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"number of box per image including image with no finding = {plot_groupby_info(train_df)}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"total number of row in train dataset {len(train_df)}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"number of class = {len(train_df['class_name'].unique())}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.class_name.value_counts()/len(train_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.class_name.value_counts()\\\n    .plot(kind='bar',\n          title='class_name',\n          figsize=(12, 4),\n          color=colorpal[0])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Check size of bounding box"},{"metadata":{"trusted":true},"cell_type":"code","source":"abnormal_df = train_df[train_df.class_name!= 'No finding']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"abnormal_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"abnormal_df['w'] = abnormal_df['x_max'].copy() - abnormal_df['x_min'].copy()\nabnormal_df['h'] = abnormal_df['y_max'] - abnormal_df['y_min']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"abnormal_df['area'] = abnormal_df['w']*abnormal_df['h']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"abnormal_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Average number of class_name and bboxes in an abnormal image\n## From the line below there are approximately (3.5 class_name) per abnormal image\n## AND 8.2 abnormal bboxes per abnormal image\n### Hence one image can contain multiple box of different type"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"number of box in train dataset/ abnormal df = {len(abnormal_df)} boxes\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"number of box per image without image with no finding = {plot_groupby_info(abnormal_df)}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"number of average box per abnormal image = {36096/4394}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 5))\nsns.distplot(abnormal_df['area'].value_counts(),\n             bins=15,\n             color=colorpal[1])\nax.set_title('Distribution bounding box sizes')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## helper function to organise dicom data and read image from path"},{"metadata":{"trusted":true},"cell_type":"code","source":"def window_imgs_from_dicom(id, dirname):\n    \n    record = {\n        'ID': id,\n#         'labels': ' '.join(labels),\n#         'n_label': len(labels),\n    }\n    \n    \n    path = '%s/%s.dicom' % (dirname, id)\n    dicom = pydicom.dcmread(path)\n    record.update(misc.get_dicom_raw(dicom))\n    \n    raw = dicom.pixel_array\n    try:\n        slope = float(record['RescaleSlope'])\n        intercept = float(record['RescaleIntercept'])\n\n        center = misc.get_dicom_value(record['WindowCenter'])\n        width = misc.get_dicom_value(record['WindowWidth'])\n\n        bits= record['BitsStored']\n        pixel = record['PixelRepresentation']\n\n#         print(center, width, bits, pixel)\n        image = misc.rescale_image(raw, slope, intercept, bits, pixel)\n\n        doctor = misc.apply_window(image, center, width)\n        brain = misc.apply_window(image, 40, 80)\n        return raw, image, doctor, brain, record\n    except:\n        return raw,raw,raw,raw, record\n    \n    \ndef create_record(id, dirname):\n\n    raw, image, doctor, brain, record = window_imgs_from_dicom(id, dirname)\n\n    record.update({\n        'raw_max': raw.max(),\n        'raw_min': raw.min(),\n        'raw_mean': raw.mean(),\n        'raw_diff': raw.max() - raw.min(),\n        'doctor_max': doctor.max(),\n        'doctor_min': doctor.min(),\n        'doctor_mean': doctor.mean(),\n        'doctor_diff': doctor.max() - doctor.min(),\n        'brain_max': brain.max(),\n        'brain_min': brain.min(),\n        'brain_mean': brain.mean(),\n        'brain_diff': brain.max() - brain.min(),\n        'brain_ratio': misc.get_windowed_ratio(image, 40, 80),\n    })\n    return record\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Scan has different size","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# ../input/vinbigdata-chest-xray-abnormalities-detection/train/000434271f63a053c4128a0ba6352c7f.dicom\n# ../input/vinbigdata-chest-xray-abnormalities-detection/train/0059d21bef1793fa9522e4ec8cae1a1a.dicom\nraw, image, doctor, brain, record = window_imgs_from_dicom('000434271f63a053c4128a0ba6352c7f',TRAIN_DIR)\nprint(f\"shape of individual scan = {raw.shape}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# ../input/vinbigdata-chest-xray-abnormalities-detection/train/000434271f63a053c4128a0ba6352c7f.dicom\n# ../input/vinbigdata-chest-xray-abnormalities-detection/train/0059d21bef1793fa9522e4ec8cae1a1a.dicom\nraw, image, doctor, brain, record = window_imgs_from_dicom('0059d21bef1793fa9522e4ec8cae1a1a',TRAIN_DIR)\nprint(f\"shape of individual scan = {raw.shape}\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## more detail about abnormal"},{"metadata":{"trusted":true},"cell_type":"code","source":"abnormal_df.describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Plot sample image of different class"},{"metadata":{"trusted":true},"cell_type":"code","source":"def imshow(img):\n    img = img / 2 + 0.5     # unnormalize\n    npimg = img.numpy()\n    plt.imshow(np.transpose(npimg, (1, 2, 0)))\n    plt.show()\n\n\n# get some random training images\n# dataiter = iter(trainloader)\n# images, labels = dataiter.next()\n\n# show images\n# imshow(torchvision.utils.make_grid(images))\n# print labels\n# print(' '.join('%5s' % classes[labels[j]] for j in range(4)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"record = create_record('000434271f63a053c4128a0ba6352c7f', TRAIN_DIR)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"windows_img = window_imgs_from_dicom('50a418190bc3fb1ef1633bf9678929b3', TRAIN_DIR)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_sample_images(df):\n    classes = df['class_name'].unique()\n    sample = []\n    for i in classes:\n        image_id = df[df.class_name == i].iloc[0].image_id\n#         print(image_id, i)\n        eg = namedtuple('id_img_pair',['id','image_id', 'window_images'])  \n        image_list = window_imgs_from_dicom(image_id, TRAIN_DIR)\n        \n        raw, image, doctor, brain, _ = image_list \n        sample.append(eg(i, image_id, [raw, image, doctor, brain]))\n    return sample","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"samples = get_sample_images(train_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Plot sample image of every class \n## (15 class as 15 rows)\n## (4 columns as 4 type of windows)"},{"metadata":{"trusted":true},"cell_type":"code","source":"import operator\nfrom functools import reduce #python 3\n\n\ndef imshow(img):\n#     img = img / 2 + 0.5     # unnormalize\n    npimg = img.numpy()\n    plt.imshow(np.transpose(npimg, (1, 2, 0)))\n    plt.show()\n\n# imshow(torchvision.utils.make_grid(torch.tensor(images_)))\n\ndef plot_img_list(img_list):\n    # settings\n    h, w = 10, 10        # for raster image\n    nrows, ncols = 1, 4  # array of sub-plots\n    figsize = [6, 8]     # figure size, inches\n\n    # prep (x,y) for extra plotting on selected sub-plots\n    xs = np.linspace(0, 2*np.pi, 60)  # from 0 to 2pi\n    ys = np.abs(np.sin(xs))           # absolute of sine\n\n    # create figure (fig), and array of axes (ax)\n    fig, ax = plt.subplots(nrows=nrows, ncols=ncols, figsize=figsize)\n#     img_list = sample_img_list[0:4]\n    # plot simple raster image on each sub-plot\n    for i, axi in enumerate(ax.flat):\n        # i runs from 0 to (nrows*ncols-1)\n        # axi is equivalent with ax[rowid][colid]\n        img = img_list[i]\n\n        axi.imshow(img, cmap=plt.cm.bone)\n        # get indices of row/column\n        rowid = i // ncols\n        colid = i % ncols\n        # write row/col indices as axes' title for identification\n        axi.set_title(\"Row:\"+str(rowid)+\", Col:\"+str(colid))\n\n    # one can access the axes by ax[row_id][col_id]\n    # do additional plotting on ax[row_id][col_id] of your choice\n\n    plt.tight_layout(True)\n    plt.show()\n\ndef plot_image_windows(df):\n    samples = get_sample_images(df)\n    \n    # settings\n    h, w = 5, 5        # for raster image\n    \n    figsize = [30, 30]     # figure size, inches\n    \n    ncols = 4\n    assert ncols == len(samples[0].window_images), 'incorrect number of window images'\n    \n    nrows = 15\n    assert nrows == len(samples), 'incorrect number of class in samples'\n    img_list = [ sample.window_images for sample in samples ]\n    # flatten list of list of images   \n    \n    img_list = reduce(operator.concat, img_list)\n    assert len(img_list) == ncols*nrows, 'incorrect number of class in samples'\n    \n\n    # prep (x,y) for extra plotting on selected sub-plots\n    xs = np.linspace(0, 2*np.pi, 60)  # from 0 to 2pi\n    ys = np.abs(np.sin(xs))           # absolute of sine\n\n    # create figure (fig), and array of axes (ax)\n    fig, ax = plt.subplots(nrows=nrows, ncols=ncols, figsize=figsize)\n    # plot simple raster image on each sub-plot\n    for i, axi in enumerate(ax.flat):\n        # i runs from 0 to (nrows*ncols-1)\n        # axi is equivalent with ax[rowid][colid]\n        img = img_list[i]\n\n#         axi.imshow(img, cmap=plt.cm.bone)\n        axi.imshow(img, cmap='gray')\n        # get indices of row/column\n        rowid = i // ncols\n        colid = i % ncols\n        # write row/col indices as axes' title for identification\n#         axi.set_title(\"Row:\"+str(rowid)+\", Col:\"+str(colid))\n    \n    # one can access the axes by ax[row_id][col_id]\n    # do additional plotting on ax[row_id][col_id] of your choice\n\n    plt.tight_layout(True)\n    plt.show()\n    return img_list\nsample_img_list = plot_image_windows(train_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Upvote if you find this useful XD"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}