{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":24800,"databundleVersionId":1831594,"sourceType":"competition"},{"sourceId":1977477,"sourceType":"datasetVersion","datasetId":1181773}],"dockerImageVersionId":30060,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Chest X-Ray Classifcation and Localization using CNN**","metadata":{}},{"cell_type":"markdown","source":"Lets get all the packages and libraries taken care of","metadata":{}},{"cell_type":"code","source":"!pip install pydicom\n!pip install Pillow\nimport os # for perfoming  directory operations \nimport glob # For transforming or changing photo formats\nfrom collections import defaultdict # for creating default dictionary array\nimport pickle  #The pickle module is used for implementing binary protocols for serializing and de-serializing a Python object structure.\n                #Source:https://www.geeksforgeeks.org/pickle-python-object-serialization/\nimport pydicom  # For reading dicom files;It lets you read, modify and write DICOM data in an easy \"pythonic\" way\nimport matplotlib.pyplot as plt ## For plotting images\nimport matplotlib.patches as patches \nfrom PIL import Image   ##\nimport numpy as np\nimport pandas as pd # for some simple data analysis (right now, just to load in the labels data and quickly reference it)\nimport tensorflow as tf\nimport seaborn as sns\nfrom PIL import Image\nfrom tensorflow.keras.layers import GlobalAveragePooling2D ,Input, Flatten , Dense , Dropout , Concatenate ,Activation\nfrom tensorflow.keras.models import Model , Sequential\nfrom tensorflow.keras.applications import mobilenet_v2\nimport shutil\nfrom glob import glob\nimport keras\nfrom keras import backend as K\nfrom keras.layers.core import Dense\nfrom keras.optimizers import Adam\nfrom keras.metrics import categorical_crossentropy\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Model\nfrom keras.applications import imagenet_utils\nfrom sklearn.metrics import confusion_matrix\nimport itertools","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Let me create an array list or data directory to store the Dicom images\n## data_dir=\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/\"\ndata_dir=\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/\"\nTrain_Dir=os.path.join(data_dir,'train')\nTest_Dir=os.path.join(data_dir,'test')\n\nTrain_Dir_Files=[os.path.join(Train_Dir,image_id) for image_id in os.listdir(Train_Dir)]\nTest_Dir_Files=[os.path.join(Test_Dir,image_id) for image_id in os.listdir(Test_Dir)]\n\nprint(\"The total number of train dicom files :\",len(Train_Dir_Files))\nprint(\"The total number of test dicom files :\",len(Test_Dir_Files))\npd.reset_option('max_colwidth')","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"I have pointed the path to my data storage  in Kaggle API. I have identified my data file as data_dir and used the 'os'library to perfom directory operations like creating an \narray list of these dicom images for both the training images and test images.\nAs i progress through my study, I would like to iterate through the list of train images or test images.\nHaving created my two array of lists 'Train_Dir_Files' and 'Test_Dir_Files', it will be easy for me to point to them or reference them.","metadata":{}},{"cell_type":"code","source":"len(Train_Dir_Files[14999])## Let me get the length of the Train_Dir list for file number 14999","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"  ??Train_Dir_Files","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(Train_Dir_Files)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(Test_Dir_Files[2999])## Let me get the length of the Train_Dir list for file number 2999","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nlen(Test_Dir_Files)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##Source:https://www.geeksforgeeks.org/iterate-over-a-list-in-python/\n# Using for loop\nfor i in Train_Dir_Files:\n    print(i)\n   ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"code","source":"## Source: https://www.youtube.com/watch?v=bCYX6wT0U7s \ndef get_train_generator(df, image_dir, x_col, y_cols, shuffle=True, batch_size=8, seed=1, target_w = 320, target_h = 320):\n     \n    print(\"getting train generator...\") \n    # normalize images\n    image_generator = ImageDataGenerator(\n        samplewise_center=True,\n        samplewise_std_normalization= True)\n    \n    # flow from directory with specified batch size\n    # and target image size\n    generator = image_generator.flow_from_dataframe(\n            dataframe=df,\n            directory=image_dir,\n            x_col=x_col,\n            y_col=y_cols,\n            class_mode=\"raw\",\n            batch_size=batch_size,\n            shuffle=shuffle,\n            seed=seed,\n            target_size=(target_w,target_h))\n    \n    return generator","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#### Source: https://www.youtube.com/watch?v=bCYX6wT0U7s \ndef get_test_and_valid_generator(valid_df, test_df, train_df, image_dir, x_col, y_cols, sample_size=100, batch_size=8, seed=1, target_w = 320, target_h = 320):\n    \n    print(\"getting train and valid generators...\")\n    # get generator to sample dataset\n    raw_train_generator = ImageDataGenerator().flow_from_dataframe(\n        dataframe=train_df, \n        directory=IMAGE_DIR, \n        x_col=\"image_id\", \n        y_col=labels, \n        class_mode=\"raw\", \n        batch_size=sample_size, \n        shuffle=True, \n        target_size=(target_w, target_h))\n    \n    # get data sample\n    batch = raw_train_generator.next()\n    data_sample = batch[0]\n\n    # use sample to fit mean and std for test set generator\n    image_generator = ImageDataGenerator(\n        featurewise_center=True,\n        featurewise_std_normalization= True)\n    \n    # fit generator to sample from training data\n    image_generator.fit(data_sample)\n\n    # get test generator\n    valid_generator = image_generator.flow_from_dataframe(\n            dataframe=valid_df,\n            directory=image_dir,\n            x_col=x_col,\n            y_col=y_cols,\n            class_mode=\"raw\",\n            batch_size=batch_size,\n            shuffle=False,\n            seed=seed,\n            target_size=(target_w,target_h))\n\n    test_generator = image_generator.flow_from_dataframe(\n            dataframe=test_df,\n            directory=image_dir,\n            x_col=x_col,\n            y_col=y_cols,\n            class_mode=\"raw\",\n            batch_size=batch_size,\n            shuffle=False,\n            seed=seed,\n            target_size=(target_w,target_h))\n    return valid_generator, test_generator\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Getting the datasets to work with\ntrain_df = pd.read_csv(\"../input/vinbigdata-chest-xray-abnormalities-detection/train.csv\")\ntest_df = pd.read_csv(\"../input/test-imgcsv/test_img.csv\")\nsample_df = pd.read_csv(\"../input/vinbigdata-chest-xray-abnormalities-detection/sample_submission.csv\")\ntrain_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The train_df file above is a summary of the findings of the images and their estimated sizes","metadata":{"trusted":true}},{"cell_type":"code","source":"Train_Dir_Files[:10]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Test_Dir_Files[:10]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##Source:https://www.geeksforgeeks.org/how-to-display-an-image-in-grayscale-in-matplotlib/\nfrom matplotlib.colors import NoNorm\nds = pydicom.dcmread(Train_Dir_Files[13999])\n#plt.imshow(ds.pixel_array,cmap='gray',norm=NoNorm())\n#ds = dicom.dcmread(Train_Dir_Files[13999])\nplt.imshow(ds.pixel_array,cmap='gray', vmin=250, vmax=2500)\n#,style=\"padding-right:50px;\", src=\"downsize_100k_v1.jpeg\", width=500, height=500)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The plt.imshow from above can help me read through all the images of my choice  by specifying the image number from the list of the 15,000 images in my list.","metadata":{"trusted":true}},{"cell_type":"code","source":"## To get more information or help learn about plt.imshow function/method\n##??plt.imshow","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Am writing a function to display the image size statistics for my images going foward\ndef get_size_statistics(ds):\n  heights = []\n  widths = []\n  for img in os.listdir(ds): \n    path = os.path.join(ds, img)\n    data = np.array(Image.open(path)) #PIL Image library\n    heights.append(data.shape[0])\n    widths.append(data.shape[1])\n  avg_height = sum(heights) / len(heights)\n  avg_width = sum(widths) / len(widths)\n  print(\"Average Height: \" + str(avg_height))\n  print(\"Max Height: \" + str(max(heights)))\n  print(\"Min Height: \" + str(min(heights)))\n  print('\\n')\n  print(\"Average Width: \" + str(avg_width))\n  print(\"Max Width: \" + str(max(widths)))\n  print(\"Min Width: \" + str(min(widths)))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Let me define my image settings\ndef window_image(img, window_center,window_width, intercept, slope):\n\n    img = (img*slope +intercept)\n    img_min = window_center - window_width//2\n    img_max = window_center + window_width//2\n    img[img<img_min] = img_min\n    img[img>img_max] = img_max\n    return img \n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Am defining the dicom fields and the windowing information metadata \ndef get_first_of_dicom_field_as_int(x):\n    #get x[0] as in int is x is a 'pydicom.multival.MultiValue', otherwise get int(x)\n    if type(x) == pydicom.multival.MultiValue:\n        return int(x[0])\n    else:\n        return int(x)\n\ndef get_windowing(data):\n    dicom_fields = [data[('0028','1050')].value, #window center\n                    data[('0028','1051')].value, #window width\n                    data[('0028','1052')].value, #intercept\n                    data[('0028','1053')].value] #slope\n    return [get_first_of_dicom_field_as_int(x) for x in dicom_fields]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport matplotlib.pyplot as plt\ncase = 13999\n\ndata = pydicom.dcmread(Train_Dir_Files[case])\n\n#print(data)\nwindow_center , window_width, intercept, slope = get_windowing(data)\n\n\n#Am displaying the same image from above ; with more details and in a grid\nimg = pydicom.read_file(Train_Dir_Files[case]).pixel_array\n\nimg = window_image(img, window_center, window_width, intercept, slope)\nplt.imshow(img, cmap=plt.cm.bone)\nplt.grid(True)\n\nprint(data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### let's get the first 20  images and their classification from the train.csv file\ntrain_df=pd.read_csv('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv', index_col=0)\nprint(\"Training DataFrame :\",format(len(train_df)),\"entries\")\ntrain_df.head(20)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## lets clean the data and replace zeros  for 'na' and 'nan'\ntrain_df.fillna(0)\ntrain_df.replace(np.nan,0)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Let me plot how my data is distributed based on class\nsns.distplot(train_df['class_id']);","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Cleaning of data and removing the 'na' and 'nan'","metadata":{}},{"cell_type":"code","source":"### Cleaning**** of data and removing the 'na' and 'nan'\ndf = train_df.fillna(0)\ndf = train_df.replace(np.nan,0)\ndf1 = df.replace(np.nan, 0)\ndf1.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.distplot(df1['x_min']);","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.distplot(df1['y_min']);","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.distplot(train_df['x_max']);","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.distplot(train_df['y_max']);","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#box plot on Radiologist interpretation vs classification ID of abnormality in Chest X-Rays\nvar = 'rad_id'\ndata = pd.concat([df1['class_id'], df1[var]], axis=1)\nf, ax = plt.subplots(figsize=(16, 10))\nfig = sns.boxplot(x=var, y='class_id', data=data)\nfig.axis(ymin=0, ymax=14);","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\nRadilogist R8, R9 and R10 have done a good job in interpretation of the Chest Images as compared to the rest","metadata":{"trusted":true}},{"cell_type":"code","source":"ds.PixelData[:200]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Because of the complexity in interpreting PixelData, pydicom provides an easy way to get it in a convenient form: pixel_array which returns a numpy.ndarray containing the pixel data:","metadata":{}},{"cell_type":"code","source":"ds.pixel_array, ds.pixel_array.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ds","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##Source: https://www.kaggle.com/zainahmad/chest-x-ray-analysis-using-deep-learning\nfig , ax  = plt.subplots(figsize=(24,12))\nsns.countplot(df1['class_name'] , ax=ax)\nax.set_title('Classification of Chest X-ray Images')\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df1['class_name'].value_counts()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The summary table from above provides the count of all the abnormalities identified in the CXR","metadata":{}},{"cell_type":"code","source":"df1.columns","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df1.groupby","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Another fravor  of cleaning data  by replacing 'na' and 'nan' using numpy library\nimport numpy as np\ndf1 = df1.replace(np.nan,0)\nfor c in df1.columns[1:]:\n    df1[c] = df1[c].astype(str)\n    \ndf1.tail(20)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"class_dist = { i : df1[i].sum() for i in df1.columns[1:]}\ncl = list(class_dist.keys())\nfreq = list(class_dist.values())\nfig , ax = plt.subplots(figsize=(24,15))\nsns.barplot(x=freq , y = class_id , ax=ax)\nax.set_title('Classification of Images')\nfig.show(class_dist)","metadata":{"trusted":true}},{"cell_type":"code","source":"df_inf = df.loc[df['class_id'] == 1]\nprint(df_inf.shape)\ndf_inf.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_inf = df.loc[df['class_id'] == 14]\nprint(df_inf.shape)\ndf_inf.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_inf = df.loc[df['class_id'] == 3]\nprint(df_inf.shape)\ndf_inf.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_inf = df.loc[df['class_id'] == 0]\nprint(df_inf.shape)\ndf_inf.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Source: https://docs.fast.ai/tutorial.medical_imaging.html\nFor more information on what dicom images are; follow the link above","metadata":{}},{"cell_type":"markdown","source":"# Section 2:  Data Processing|EDA\n\nTo analyze our dataset, we load the paths to the DICOM files with the get_dicom_files function. When calling the function, we append train/ to the pneumothorax_source path to choose the folder where the DICOM files are located. We store the path to each DICOM file in the items list.","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport pydicom\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nfrom skimage import exposure\nimport cv2\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install ensemble-boxes\n!pip install tensorflow_io","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset_dir = '../input/vinbigdata-chest-xray-abnormalities-detection'","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def dicom2array(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to\n    # transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n        \n    \ndef plot_img(img, size=(7, 7), is_rgb=True, title=\"\", cmap='gray'):\n    plt.figure(figsize=size)\n    plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n\n\ndef plot_imgs(imgs, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500,500)):\n    rows = len(imgs)//cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows, cols, i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n    \ndef draw_bboxes(img, boxes, thickness=10, color=(255, 0, 0), img_size=(500,500)):\n    img_copy = img.copy()\n    if len(img_copy.shape) == 2:\n        img_copy = np.stack([img_copy, img_copy, img_copy], axis=-1)\n    for box in boxes:\n        img_copy = cv2.rectangle(\n            img_copy,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness)\n    if img_size is not None:\n        img_copy = cv2.resize(img_copy, img_size)\n    return img_copy","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Let me look at the 12 Images from my dataset_dir\ndicom_paths = glob(f'{dataset_dir}/train/*.dicom')\nimgs = [dicom2array(path) for path in dicom_paths[:12]]\nplot_imgs(imgs)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Bring about equalizer exposure\nimgs = [exposure.equalize_hist(img) for img in imgs]\nplot_imgs(imgs)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The concept of histogram equilization improves the contrast in images and thus brings about clarity of purpose  \n#https://opencv-python-tutroals.readthedocs.io/en/latest/py_tutorials/py_imgproc/py_histograms/py_histogram_equalization/py_histogram_equalization.html","metadata":{"trusted":true}},{"cell_type":"code","source":"from bokeh.plotting import figure as bokeh_figure\nfrom bokeh.io import output_notebook, show, output_file\nfrom bokeh.models import ColumnDataSource, HoverTool, Panel\nfrom bokeh.models.widgets import Tabs\nimport pandas as pd\nfrom PIL import Image\nfrom sklearn import preprocessing\nimport random\nfrom random import randint","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(finding_df['class_name'] == 'No finding').unique()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Plotting/Localizing of boxes in the X-Ray Images\nWe are going to plot boxes in the X-Ray images to show the exact place where there is an abnormality","metadata":{}},{"cell_type":"code","source":"##Source: https://www.kaggle.com/adrihasfi/vinbigdata-chest-xray-abnormality-eda\nimgs = []\nimg_ids = finding_df['image_id'].values\nclass_ids = finding_df['class_id'].unique()\n\n# map label_id to specify color\nlabel2color = {class_id:[randint(0,255) for i in range(3)] for class_id in class_ids}\nthickness = 3\nscale = 5\n\n\nfor i in range(8):\n    img_id = random.choice(img_ids)\n    img_path = f'{dataset_dir}/train/{img_id}.dicom'\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    boxes = finding_df.loc[finding_df['image_id'] == img_id, ['x_min', 'y_min', 'x_max', 'y_max']].values/scale\n    labels = finding_df.loc[finding_df['image_id'] == img_id, ['class_id']].values.squeeze()\n    \n    for label_id, box in zip(labels, boxes):\n        color = label2color[label_id]\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n    )\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The boxes with the same color which overlaps indicates that the radiologist had similar predictions about the intepretation of the images","metadata":{}},{"cell_type":"code","source":"##Histogram  plots\ndef hist_ploting(dataframe, column, color=[\"#94c3428\", \"#ea2e31\"], bins=30, title=\"\", value_range=None):\n   ##histogram\n\n    hist, edges = np.histogram(dataframe[column], bins=bins, range=value_range)\n    hist_frame = pd.DataFrame({\n        column: hist,\n        \"left\": edges[:-1],\n        \"right\": edges[1:]\n    })\n    hist_frame[\"interval\"] = [\"%d to %d\" %\n                              (left, right) for left, right in zip(edges[:-1], edges[1:])]\n    src = ColumnDataSource(hist_frame)\n    plot = bokeh_figure(\n        plot_height=400, plot_width=600,\n        title=title, x_axis_label=column,\n        y_axis_label=\"Count\"\n    )\n    plot.quad(\n        bottom=0, top=column, left=\"left\", right=\"right\",\n        source=src, fill_color=color[0], line_color=\"#358388\",\n        fill_alpha=0.7, hover_fill_alpha=0.7,\n        hover_fill_color=color[1]\n    )\n    hover = HoverTool(\n        tooltips=[(\"Interval\", \"@interval\"), (\"Count\", str(f\"@{column}\"))]\n    )\n    plot.add_tools(hover)\n    output_notebook()\n    show(plot)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import plotly.graph_objects as go","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_distribution_classes(x_values, y_values):\n    \n    colors = ['rgb(26, 118, 255)',] * 15\n    colors[0] = 'lightslategray'\n\n    fig = go.Figure(data=[go.Bar(\n        x=x_values, \n        y=y_values,\n        text=y_values,\n        marker_color=colors\n    )])\n\n    fig.update_layout(height=600, width=900, title_text=\"Distribution of radiographic observations\")\n    fig.update_xaxes(type=\"category\")\n\n    fig.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"indexes = train_df.class_name.unique()\ncounts = train_df.class_name.value_counts()\n\nsorted_dict = dict(zip(indexes, counts))\nsorted_dict = {k: v for k, v in sorted(sorted_dict.items(), key=lambda item: item[1], reverse = True)}\n\nx = list(sorted_dict.keys())\ny = list(sorted_dict.values())\n\nplot_distribution_classes(x, y)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_distribution_radiologist_obs(x_values, y_values):\n    \n    colors = ['lightslategray',] * 17\n    colors[0] = 'crimson'\n    colors[1] = 'crimson'\n    colors[2] = 'crimson'\n    \n    fig = go.Figure(data=[go.Bar(\n        x=x_values, \n        y=y_values,\n        text=y_values,\n        marker_color=colors\n    )])\n\n    fig.update_layout(height=600, width=900, title_text=\"Distribution of radiologist observations\")\n    fig.update_xaxes(type=\"category\")\n\n    fig.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#The graph below demonstrates the distribution of Radiologists and their classification\nindexes = train_df[['rad_id', 'image_id']].groupby(['rad_id']).agg(['count']).index\ncounts = train_df[['rad_id', 'image_id']].groupby(['rad_id']).agg(['count']).values.ravel()\n\nsorted_dict = dict(zip(indexes, counts))\nsorted_dict = {k: v for k, v in sorted(sorted_dict.items(), key=lambda item: item[1], reverse = True)}\n\nx = list(sorted_dict.keys())\ny = list(sorted_dict.values())\n\nplot_distribution_radiologist_obs(x, y)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The radiologists with high ranking contributes more to the intepretation of Chest X-Ray images","metadata":{}},{"cell_type":"code","source":"train_df.rad_id","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.class_name","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#Source:Project: CNNArt   Author: thomaskuestner  \ndef create_DICOM_Array(PathDicom):\n    filenames_list = []\n\n    file_list = os.listdir(PathDicom)\n    for file in file_list:\n        filenames_list.append(PathDicom + file)\n    datasets = [dicom.read_file(f) \\\n                for f in filenames_list]\n    try:\n        voxel_ndarray, _ = dicom_numpy.combine_slices(datasets)\n        voxel_ndarray = voxel_ndarray.astype(float)\n        voxel_ndarray = np.swapaxes(voxel_ndarray, 0, 1)\n        print(voxel_ndarray.dtype)\n        # voxel_ndarray = voxel_ndarray[:-1:]\n        # print(voxel_ndarray.shape)\n    except dicom_numpy.DicomImportException:\n        # invalid DICOM data\n        raise\n\n    print(voxel_ndarray.shape)\n    return voxel_ndarray ","metadata":{}},{"cell_type":"code","source":"## Lets get the first 10 list of dicom train images \nTrain_Dir_Files[:10]","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Let's point our file reader to the path of the file\ndata_dir = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/\"\nima = os.listdir(data_dir)\nlabels_df = pd.read_csv('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/sample_submission.csv', index_col=0)\n\nlabels_df.tail(20)","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_df['PredictionString'].value_counts()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ima[1:3]","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image # used for loading images\nimport numpy as np\nimport os # used for navigating to image path\nimport imageio # used for writing images\nnaming_dict = {} # id: breed\nf = open(\"../input/test-imgcsv/test_img.csv\", \"r\")\nfileContents = f.read()\nfileContents = fileContents.split('\\n')\nfor i in range(len(fileContents)-1):\n  fileContents[i] = fileContents[i].split(',')\n  naming_dict[fileContents[i][0]] = fileContents[i][1]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## This examples illustrates how to work with sequences.\n\n\n\nfrom pydicom.sequence import Sequence\nfrom pydicom.dataset import Dataset\n\n# create to toy datasets\nblock_ds1 = Dataset()\nblock_ds1.BlockType = \"APERTURE\"\nblock_ds1.BlockName = \"Block1\"\n\nblock_ds2 = Dataset()\nblock_ds2.BlockType = \"APERTURE\"\nblock_ds2.BlockName = \"Block2\"\n\nbeam = Dataset()\n# note that you should add beam data elements like BeamName, etc; these are\n# skipped in this example\nplan_ds = Dataset()\n# starting from scratch since we did not read a file\nplan_ds.BeamSequence = Sequence([beam])\nplan_ds.BeamSequence[0].BlockSequence = Sequence([block_ds1, block_ds2])\nplan_ds.BeamSequence[0].NumberOfBlocks = 2\n\nbeam0 = plan_ds.BeamSequence[0]\nprint('Number of blocks: {}'.format(beam0.BlockSequence))\n\n# create a new data set\nblock_ds3 = Dataset()\n# add data elements to it as above and don't forget to update Number of Blocks\n# data element\nbeam0.BlockSequence.append(block_ds3)\ndel plan_ds.BeamSequence[0].BlockSequence[1]\n","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename, ))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_resized=pd.read_csv(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/sample_submission.csv\")\ntrain_resized.rename(columns={'dim0':'height','dim1':'width'},inplace=True)\nprint(\"Metadata for Resized Training Images :\",format(len(train_resized)),\"entries\")\ntrain_resized.head()","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_PX_SIZE = 224\ndef resize(img_dcm):\n    return cv2.resize(np.array(img_dcm.pixel_array, (IMG_PX_SIZE,IMG_PX_SIZE)))","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#%reload_ext signature\n%matplotlib inline\n\nimport numpy as np\n#import dicom\nimport os\nimport matplotlib.pyplot as plt\nfrom glob import glob\nfrom mpl_toolkits.mplot3d.art3d import Poly3DCollection\nimport scipy.ndimage\nfrom skimage import morphology\nfrom skimage import measure\nfrom skimage.transform import resize\nfrom sklearn.cluster import KMeans\nfrom plotly import __version__\nfrom plotly.offline import download_plotlyjs, init_notebook_mode, plot, iplot\nfrom plotly.tools import FigureFactory as FF\nfrom plotly.graph_objs import *\ninit_notebook_mode(connected=True) ","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_path = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/\"\noutput_path = working_path = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train_data/\"\ng = glob(data_path + '/*.dicom')\n\n# Print out the first 5 file names to verify we're in the right folder.\nprint (\"Total of %d DICOM images.\\nFirst 5 filenames:\" % len(g))\n#print '\\n'.join(g[:5])","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def getBoxes(img_id):\n    boxes=pd.DataFrame(columns=['class_name','class_id','rad_id','bbox'])\n    for row in train[train['image_id']==img_id].to_numpy():\n        \n        # drop the 'No Finding' classes\n        if not np.isnan(row[4]):\n            boxes=boxes.append({'class_name':row[1],'class_id':row[2],'rad_id':row[3],'bbox':row[4:]},ignore_index=True)\n\n    return boxes\n    \ndef getImage_id(path):\n    return path[61:93]\n\ndef getcolour(name):\n    out=classes[classes['class_name']==name]['color_scheme']\n    return list(out)[0]\n    \ndef drawBoxes(path,reduced,opacity=0.1):\n    font=cv2.FONT_HERSHEY_SIMPLEX;\n    \n    # convert ndarray to grayscale with 3 channels\n    img=cv2.cvtColor(read_xray(path),cv2.COLOR_GRAY2RGB)\n    if reduced:\n        boxes=reduce_Boxes(getImage_id(path))\n    else:\n        boxes=getBoxes(getImage_id(path))\n    \n    # add annotations and bounding boxes\n    for row in boxes.to_numpy():\n        rect = np.uint8(np.ones((int(row[3][3]-row[3][1]), int(row[3][2]-row[3][0]), 3))*getcolour(row[0]))\n        img=cv2.rectangle(img,\n                (int(row[3][0]), int(row[3][1])),\n                (int(row[3][2]), int(row[3][3])),\n                getcolour(row[0]), 5)\n        sub_combo = cv2.addWeighted(img[int(row[3][1]):int(row[3][3]),int(row[3][0]):int(row[3][2]),:], 1-opacity, rect, opacity, 1.0)    \n        img[int(row[3][1]):int(row[3][3]),int(row[3][0]):int(row[3][2]),:] = sub_combo\n        img=cv2.putText(img,row[0],\n                (int(row[3][0]), int(row[3][1])-12),\n                 font,2,getcolour(row[0]),4)\n    return img","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"I give credit to my fellow Kaggler for the data pre-processing code; It was very simple and more easy to work with to customice it to serve my needs \nSource: Anton Morenov\n   https://www.kaggle.com/morenovanton/vinbigdata-chest-x-ray-convert-resize-image-to-png","metadata":{}},{"cell_type":"markdown","source":"# **Section 3: Data Preprocessing, Analysis & Visualization**   \n\nSome of my data contain NaNs; I need to get this taken care of, but I need to understand my data first before I make a decision on how to clean the data without messing \nit.\nLet me find out by checking what my sample submission is all about","metadata":{}},{"cell_type":"markdown","source":"Some other activities involving pre-processing includes the conversion or change of image formats as demonstrated in the next slides\nConverting DICOM files to png images. \nFrom https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image","metadata":{}},{"cell_type":"code","source":"!pip install pypng","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import png","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/\"\ntest_dir = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/test/\"\n\ntrain_images = os.listdir(train_dir)\ntest_images = os.listdir(test_dir)\n\ntrain_path_image = [train_dir + image for image in train_images]\ntest_path_image = [test_dir + image for image in test_images]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filetrain = './trainpng/' \n\nif not os.path.exists(os.path.dirname(filetrain)):\n    os.makedirs(filetrain)\n    \nfiletest = './testpng/' \n\nif not os.path.exists(os.path.dirname(filetest)):\n    os.makedirs(filetest)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filetrain + train_path_image[0].split('/')[-1].replace(\".dicom\", \".png\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_to_png(full_path_image, filename):\n    print(filename)\n    for dicomimage in tqdm(full_path_image):\n        ds = pydicom.dcmread(dicomimage)\n        img = apply_voi_lut(ds.pixel_array, ds)\n        img = resize(img, (1024, 1024), anti_aliasing=True)\n        \n        if ds.PhotometricInterpretation == \"MONOCHROME1\":\n            img = np.amax(img) - img\n            \n        img = (((img - np.min(img))/np.max(img))*255.0).astype(np.uint8) \n        \n        with open(filename + dicomimage.split('/')[-1].replace(\".dicom\", \".png\"), \"wb\") as fn:\n            \n            w = png.Writer(1024, 1024, greyscale=True)\n            w.write(fn, img)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.groupby('class_id').size())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.groupby('rad_id').size())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"convert_to_png(train_path_image[:100], filetrain)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"convert_to_png(test_path_image[:100], filetest)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!tar -zcf train.tar.gz  './trainpng'\n!tar -zcf test.tar.gz  './testpng'","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"import pydicom\ndataset = pydicom.dcmread(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv\", force=True)\nds = os.listdir(dataset)\n\n","metadata":{"trusted":true,"scrolled":true}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom ensemble_boxes import *\nfrom tqdm.notebook import tqdm\n\nimport pydicom\nfrom pydicom.tag import Tag\n\nimport tensorflow as tf\nimport tensorflow_io as tfio\n\n#from object_detection.protos.string_int_label_map_pb2 import StringIntLabelMap, StringIntLabelMapItem\n#from object_detection.dataset_tools import tf_record_creation_util\n#from object_detection.utils import dataset_util\nimport contextlib2\n\nfrom google.protobuf import text_format","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_dicom(path, max_dim):\n    image_bytes = tf.io.read_file(path)\n    image = tfio.image.decode_dicom_image(\n        image_bytes, \n        dtype = tf.uint16\n    )\n    \n    image = tf.squeeze(image, axis = 0)\n    \n    h, w, _ = image.shape\n    \n    image = tf.image.resize(\n        image, \n        (max_dim, max_dim), \n        preserve_aspect_ratio = True\n    )\n    \n    image = image - tf.reduce_min(image)\n    image = image / tf.reduce_max(image)\n    image = tf.cast(image * 255, tf.uint8)\n    \n    return image, h, w","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%matplotlib inline\n\nmax_dim = 500\ndemo_image = \"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/00053190460d56c53cc3e57321387478.dicom\"\nimage, h, w = pydicom.read_file(os.path.join(path, \"train\", demo_image), max_dim)\n\nplt.figure(figsize = (5, 5))\nplt.imshow(tf.squeeze(image), 'gray')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using image preprocessing called  \"CLAHE\" (Contrast Limited Adaptive Histogram Equalization) is similar to some extent ,\n#to some of the work we did before  about histogram equalization. It re-distributes the lightness values of the image, making patterns more visible.\ndef CLAHE(image):\n    clahe = cv2.createCLAHE(\n        clipLimit = 2., \n        tileGridSize = (100, 100)\n    )\n    \n    image = clahe.apply(image.numpy()) \n    image = tf.expand_dims(image, axis = 2)\n    \n    return image","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%matplotlib inline\n\nfig = plt.figure(figsize = (8, 8))\n\naxes = fig.add_subplot(1, 2, 1)\nplt.imshow(tf.squeeze(image), cmap = \"gray\")\naxes.set_title(\"Original\")\n\naxes = fig.add_subplot(1, 2, 2)\nimage = CLAHE(image)\nplt.imshow(tf.squeeze(image), cmap = \"gray\")\naxes.set_title(\"Post CLAHE\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##The API requires the classes to be from 1 to n and outputs 0 when no class is found. Since our labels start with 0,\n##we make unit increment to the class_id and use the new label-map.","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Creating LabelMap\ndf[\"class_id\"] = df[\"class_id\"] + 1 # Incrementing by 1\nLabelMap = df.loc[df[\"class_name\"] != \"No finding\", [\"class_name\", \"class_id\"]] # Removing the examples with no finding\nLabelMap = LabelMap.drop_duplicates().reset_index(drop = True)\nLabelMap","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Using 14 unique colors to annotate the abnormalities.\nLABEL_COLORS = [\n    (230, 25, 75), (60, 180, 75), (255, 225, 25), (0, 130, 200), (245, 130, 48), (145, 30, 180), (70, 240, 240), \n    (240, 50, 230), (210, 245, 60), (250, 190, 212), (0, 128, 128), (220, 190, 255), (170, 110, 40), (255, 250, 200), \n]\nLabelMap[\"colors\"] = LABEL_COLORS","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ds1 = pydicom.read_file(\"../input/vinbigdata-chest-xray-abnormalities-detection/train/000434271f63a053c4128a0ba6352c7f.dicom\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom pydicom import dcmread\nfrom pydicom.data import get_testdata_file\n\n# The path to a pydicom test dataset\npath = pydicom.read_file(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/000434271f63a053c4128a0ba6352c7f.dicom\")\nds = path\n# `arr` is a numpy.ndarray\narr = ds.pixel_array\n\nplt.imshow(arr, cmap=\"gray\")\nplt.show()","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.distplot(train_df['class_id']);","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Let me read some of the images to get a glimpse\nfrom pydicom import dcmread\nfrom pydicom.data import get_testdata_file\npath = pydicom.read_file(\"../input/vinbigdata-chest-xray-abnormalities-detection/train/000434271f63a053c4128a0ba6352c7f.dicom\")\nds = path\ntype(ds.PixelData)\n#<class 'bytes'>\nlen(ds.PixelData)","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Sample_submission_df = pd.read_csv('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/sample_submission.csv', index_col=0) \nSample_submission_df.head(10) # Am getting the first 10 lines of data from the train csv file","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The sample submission file gives me an idea about the predictions expected from the Chest-Xray images above\nLet me infer the information from the competion to ensure that I stay within the limits.\n\n\n","metadata":{}},{"cell_type":"markdown","source":"# Section 4: Data Modeling","metadata":{"trusted":true,"scrolled":true}},{"cell_type":"markdown","source":"### Deliverables\nUse deep learning from the tensorflow package to train the dataset.\n\n    - Explain the analysis process. \n    - Investigate prediction performance on multiple runs (experiment by varying parameters such as:\n    - numbers of layers, \n    - numbers of nodes, and so on).\n\n     - Discuss results, \n     - conclude your findings, and \n     - give insights.\n\nYou can also use Tensorflow & Deep Learning with Python. (with python 3, \n\n    - provide environment details, \n    - libraries needed, and so on) \n    - Make sure that all libraries are included and ready to run.\n\n\n## Installation of Tensorflow is the first task","metadata":{}},{"cell_type":"markdown","source":"I used the CNN model to run my training data and tested the model with my test data\nMy model contains several building blocks ;\nInput layer\nHidden layers; contain convolution with RELU layer and pooling layer\nOutput layer\nRelu- is performed with non-linear activation function that sets negative input values to zero.\nAfter completion; the pooling layer executes sampling operation\n","metadata":{}},{"cell_type":"code","source":"#pip install tensorflow","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#pip install spark_tensorflow_distributor","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\npip install spark-tensorflow-distributor","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\npip install pyspark","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Let me check that I have tensorflow version for my predictions","metadata":{}},{"cell_type":"code","source":"pip list | grep tensorflow ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pyspark\nfrom spark_tensorflow_distributor import MirroredStrategyRunner\n\n# Taken from https://github.com/tensorflow/ecosystem/tree/master/spark/spark-tensorflow-distributor#examples\ndef train():\n    import tensorflow as tf\n    import uuid\n\n    BUFFER_SIZE = 20000\n    BATCH_SIZE = 64\n\n    def make_datasets():\n        (train_images, train_labels), _ = \\\n            tf.keras.datasets.vinbigdata.load_data(path=str(uuid.uuid4())+'/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/')\n\n        dataset = tf.data.Dataset.from_tensor_slices((\n            tf.cast(vinbigdata_images[..., tf.newaxis] / 255.0, tf.float32),\n            tf.cast(vinbigdata_labels, tf.int64))\n        )\n        dataset = dataset.repeat().shuffle(BUFFER_SIZE).batch(BATCH_SIZE)\n        return dataset\n\n    def build_and_compile_cnn_model():\n        model = tf.keras.Sequential([\n            tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(28, 28, 1)),\n            tf.keras.layers.MaxPooling2D(),\n            tf.keras.layers.Flatten(),\n            tf.keras.layers.Dense(64, activation='relu'),\n            tf.keras.layers.Dense(10, activation='softmax'),\n        ])\n        model.compile(\n            loss=tf.keras.losses.sparse_categorical_crossentropy,\n            optimizer=tf.keras.optimizers.SGD(learning_rate=0.001),\n            metrics=['accuracy'],\n        )\n        return model\n    train_datasets = make_datasets()\n    options = tf.data.Options()\n    options.experimental_distribute.auto_shared_policy = tf.data.experimental.AutoSharedPolicy.DATA\n    train_datasets = train_datasets.with_options(options)\n    multi_worker_model = build_and_compile_cnn_model()\n    multi_worker_model.fit(x=train_datasets, epochs=3, steps_per_epoch=5)","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef display_images(class_id, graph_indexes = np.arange(9)):\n    \n    # Get files\n    files_index = class_id_index_examples[str(class_id)]\n    files_list = class_id_image_examples[str(class_id)]\n    \n    # define subplot\n    fig, axs = plt.subplots(3,3, figsize=(12,12))\n    for graph_index in graph_indexes:\n        \n        full_filename = files_list[graph_index]+'.dicom'\n        ds = dcmread(os.path.join(PATH, \n                                  'train',\n                                  full_filename))\n        \n\n#         axs[graph_index%3, (graph_index)//3].set_title('Label: %s \\n'%class_id,\n#                   fontsize=18)\n        axs[graph_index%3, (graph_index)//3].imshow(ds.pixel_array, cmap=plt.get_cmap('gray'))\n                  \n        if str(class_id) != '14':\n            \n            # Add rectangle\n            anchor_point, height, width = get_rectangle_parameter(train_dataframe, \n                                                                  files_index[graph_index])\n            rect = Rectangle(anchor_point, \n                                     height, \n                                     width, \n                                     edgecolor='r', \n                                     facecolor=\"none\")\n            axs[graph_index%3, (graph_index)//3].add_patch(rect)\n                     \n    # the bottom of the subplots of the figure\n    plt.subplots_adjust(bottom = 0.001)\n    plt.subplots_adjust(top = 0.99)\n    \n    # show the figure\n    plt.show()\n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import libraries\nfrom sklearn import datasets\nimport numpy as np\nimport os\nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\nfrom pandas import DataFrame\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\nimport tensorflow.compat.v2 as tf\nfrom tensorflow.keras.datasets import mnist as tfds\nimport tensorflow_datasets as tfds\ntf.enable_v2_behavior()","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Step 1: Loading data and creating an input pipeline","metadata":{}},{"cell_type":"code","source":"!pwd","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## data_dir=\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/\"\ndata_dir=\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/\"\nTrain_Dir=os.path.join(data_dir,'train')\nTest_Dir=os.path.join(data_dir,'test')\n\nTrain_Dir_Files=[os.path.join(Train_Dir,name) for name in os.listdir(Train_Dir)]\nTest_Dir_Files=[os.path.join(Test_Dir,name) for name in os.listdir(Test_Dir)]\n\nprint(\"The total number of train dicom files :\",len(Train_Dir_Files))\nprint(\"The total number of test dicom files :\",len(Test_Dir_Files))\npd.reset_option('max_colwidth')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications.densenet import DenseNet121\nfrom keras.layers import Dense, GlobalAveragePooling2D\nfrom keras.models import Model\nfrom keras import backend as K\n\nfrom keras.models import load_model\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport time \nimport tensorflow.compat.v1 as tf\n\ntf.disable_v2_behavior() \n\nstart = time.clock()\nIMG_SIZE_PX = 500\nSLICE_COUNT = 20\n\nn_classes = 14\nbatch_size = 10\n\nx = tf.placeholder('float')\ny = tf.placeholder('float')\n\nkeep_rate = 0.8","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def compute_class_freqs(labels):\n    \n    # total number of patients (rows)\n    N = labels.shape[0]\n    \n    positive_frequencies = np.sum(labels, axis=0) / N\n    negative_frequencies = 1 - positive_frequencies\n\n    return positive_frequencies, negative_frequencies","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_train_generator(df, image_dir, x_col, y_cols, shuffle=True, batch_size=8, seed=1, target_w = 320, target_h = 320):\n     \n    print(\"getting train generator...\") \n    # normalize images\n    image_generator = ImageDataGenerator(\n        samplewise_center=True,\n        samplewise_std_normalization= True)\n    \n    # flow from directory with specified batch size\n    # and target image size\n    generator = image_generator.flow_from_dataframe(\n            dataframe=df,\n            directory=image_dir,\n            x_col=x_col,\n            y_col=y_cols,\n            class_mode=\"raw\",\n            batch_size=batch_size,\n            shuffle=shuffle,\n            seed=seed,\n            target_size=(target_w,target_h))\n    \n    return generator","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_test_and_valid_generator(valid_df, test_df, train_df, image_dir, x_col, y_cols, sample_size=100, batch_size=8, seed=1, target_w = 320, target_h = 320):\n    \n    print(\"getting train and valid generators...\")\n    # get generator to sample dataset\n    raw_train_generator = ImageDataGenerator().flow_from_dataframe(\n        dataframe=train_df, \n        directory=IMAGE_DIR, \n        x_col=\"Image\", \n        y_col=labels, \n        class_mode=\"raw\", \n        batch_size=sample_size, \n        shuffle=True, \n        target_size=(target_w, target_h))\n    \n    # get data sample\n    batch = raw_train_generator.next()\n    data_sample = batch[0]\n\n    # use sample to fit mean and std for test set generator\n    image_generator = ImageDataGenerator(\n        featurewise_center=True,\n        featurewise_std_normalization= True)\n    \n    # fit generator to sample from training data\n    image_generator.fit(data_sample)\n\n    # get test generator\n    valid_generator = image_generator.flow_from_dataframe(\n            dataframe=valid_df,\n            directory=image_dir,\n            x_col=x_col,\n            y_col=y_cols,\n            class_mode=\"raw\",\n            batch_size=batch_size,\n            shuffle=False,\n            seed=seed,\n            target_size=(target_w,target_h))\n\n    test_generator = image_generator.flow_from_dataframe(\n            dataframe=test_df,\n            directory=image_dir,\n            x_col=x_col,\n            y_col=y_cols,\n            class_mode=\"raw\",\n            batch_size=batch_size,\n            shuffle=False,\n            seed=seed,\n            target_size=(target_w,target_h))\n    predicted_vals = model.predict_generator(test_generator, steps = len(test_generator))\n    return valid_generator, test_generator","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Source:https://www.kaggle.com/jiandanjinxin/first-pass-through-data-w-3d-convnet\ndef conv3d(x, W):\n    return tf.nn.conv3d(x, W, strides=[1,1,1,1,1], padding='SAME')\n\ndef maxpool3d(x):\n    #                        size of window         movement of window as you slide about\n    return tf.nn.max_pool3d(x, ksize=[1,2,2,2,1], strides=[1,2,2,2,1], padding='SAME')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convolutional_neural_network(x):\n    #                # 5 x 5 x 5 patches, 1 channel, 32 features to compute.\n    weights = {'W_conv1':tf.Variable(tf.random_normal([3,3,3,1,32])),\n               #       5 x 5 x 5 patches, 32 channels, 64 features to compute.\n               'W_conv2':tf.Variable(tf.random_normal([3,3,3,32,64])),\n               #                                  64 features\n               #'W_conv3':tf.Variable(tf.random_normal([3,3,3,64,128])),\n               \n               'W_fc':tf.Variable(tf.random_normal([54080,1024])),\n               'out':tf.Variable(tf.random_normal([1024, n_classes]))}\n\n    biases = {'b_conv1':tf.Variable(tf.random_normal([32])),\n               'b_conv2':tf.Variable(tf.random_normal([64])),\n               'b_fc':tf.Variable(tf.random_normal([1024])),\n               'out':tf.Variable(tf.random_normal([n_classes]))}\n\n\n    conv1 = tf.nn.relu(conv3d(x, weights['W_conv1']) + biases['b_conv1'])\n    conv1 = maxpool3d(conv1)\n\n\n    conv2 = tf.nn.relu(conv3d(conv1, weights['W_conv2']) + biases['b_conv2'])\n    conv2 = maxpool3d(conv2)\n    \n   # conv3 = tf.nn.relu(conv3d(conv2, weights['W_conv3']) + biases['b_conv3'])\n   # conv3 = maxpool3d(conv3)\n\n    fc = tf.reshape(conv2,[-1, 54080])\n    fc = tf.nn.relu(tf.matmul(fc, weights['W_fc'])+biases['b_fc'])\n    fc = tf.nn.dropout(fc, keep_rate)\n\n    output = tf.matmul(fc, weights['out'])+biases['out']\n\n    return output","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.layers import Dense\n\nbase_model = MobileNetV2(\n    include_top=False, \n    input_shape=(100, 100, 3),\n    weights=None\n)\n\nlayer = Dense(256, activation='relu')(base_model.output)\nout = Dense(28)(layer)\n\nmodel = Model(base_model.input, out)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_weighted_loss(pos_weights, neg_weights, epsilon=1e-7):\n   \n    def weighted_loss(y_true, y_pred):\n    \n        loss = 0.0\n\n        for i in range(len(pos_weights)):\n            # for each class, add average weighted loss for that class\n            loss_pos = -1 * K.mean(pos_weights[i] * y_true[:, i] * K.log(y_pred[:, i] + epsilon))\n            loss_neg = -1 * K.mean(neg_weights[i] * (1 - y_true[:, i]) * K.log(1 - y_pred[:, i] + epsilon))\n            loss += loss_pos + loss_neg\n        \n        return loss\n    \n        ### END CODE HERE ###\n    return weighted_loss","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#much_data = np.load('muchdata-50-50-20.npy')\nmuch_data = Train_Dir_Files\n# I will be using my generated list of train data and test data\ntrain_data = much_data[:-10]\nvalidation_data = much_data[-10:]\n\n#                            image X      image Y        image Z\nx = tf.reshape(x, shape=[-1, IMG_SIZE_PX, IMG_SIZE_PX, SLICE_COUNT, 1])\ndef train_neural_network(x):\n    prediction = convolutional_neural_network(x)\n    cost = tf.reduce_mean( tf.nn.softmax_cross_entropy_with_logits(logits=prediction,labels=y) )\n    optimizer = tf.train.AdamOptimizer(learning_rate=1e-3).minimize(cost)\n    \n    hm_epochs = 3\n    with tf.Session() as sess:\n        #sess.run(tf.initialize_all_variables())\n        sess.run(tf.global_variables_initializer())\n        \n        successful_runs = 0\n        total_runs = 0\n        \n        for epoch in range(hm_epochs):\n            epoch_loss = 0\n            for data in train_data:\n                total_runs += 1\n                try:\n                    X = data[0]\n                    Y = data[1]\n                    _, c = sess.run([optimizer, cost], feed_dict={x: X, y: Y})\n                    epoch_loss += c\n                    successful_runs += 1\n                except Exception as e:\n                    # I am passing for the sake of notebook space, but we are getting 1 shaping issue from one \n                    # input tensor. Not sure why, will have to look into it. Guessing it's\n                    # one of the depths that doesn't come to 20.\n                    pass\n                    #print(str(e))\n            \n            print('Epoch', epoch+1, 'completed out of',hm_epochs,'loss:',epoch_loss)\n\n            correct = tf.equal(tf.argmax(prediction, 1), tf.argmax(y, 1))\n            #accuracy = tf.reduce_mean(tf.cast(correct, 'float'))\n            accuracy = tf.reduce_mean(tf.cast(correct, tf.float64))\n\n            print('Accuracy:',accuracy.eval({x:[i[0] for i in validation_data], y:[i[1] for i in validation_data]}))\n            \n        print('Done. Finishing accuracy:')\n        print('Accuracy:',accuracy.eval({x:[i[0] for i in validation_data], y:[i[1] for i in validation_data]}))\n        \n        print('fitment percent:',successful_runs/total_runs)\nstring = input(x) \n# Run this locally:\n# train_neural_network(x)\ntrain_neural_network(x)\nend = time.clock()\nprint(\"The running time is %g s\" %(end-start))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Explanation about some of the arguements used above\n* The split args; splits data into  training and testing parts\n* Shuffle_files; controls whether to shuffle  the files between each epoch\n* tfds;  stores big datasets in multiple smaller files\n* data_dir;  si the location where the dataset is saved~ dfaults to tensorflow datasets\n* * builder function; compiles datasets and different datasets together.It downloads and prepares a dataset for use ","metadata":{}},{"cell_type":"code","source":"../input/vinbigdata-chest-xray-abnormalities-detection","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ds","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(ds)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"builder = tfds.builder('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/')\n# 1. Create the tfrecord files (no-op if already exists)\nbuilder.download_and_prepare()\n# 2. Load the `tf.data.Dataset`\nds = builder.as_dataset(split='train', shuffle_files=True)\nprint(ds)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Section 5: Model Testing and Evaluation","metadata":{"trusted":true,"scrolled":true}},{"cell_type":"markdown","source":"The testing pipeline is similar to the training pipeline with small differences then caching follows between the epoch\nIteration and creating of model by changing parameters; Using Sigmoid activation instead of 'relu'\nReduce and Increase layers to test \nWith reduction of layers, accuracy drops significantly from 97.4% to 94.37% Let me see what happens with increase of number of layers.\n","metadata":{"trusted":true}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# * Section 6: Inferences","metadata":{}},{"cell_type":"markdown","source":"The model was able to give a 74% prediction with consistent accurate results\nChest x-Ray classification and localization was successful and comparable to Radiologist classification.\nThe prediction/classification is faster  compared to human classification by radiologists\nThere’s promising outcome from the use of this model in Chest X-Ray prediction and classification\nWith increase of layers like the previous model. Accuracy improved , though not as much as when using the 'relu' activation. Our accuracy improved to 96.32 , but still its less than the previous 97.59%\n","metadata":{"trusted":true}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Section 7:Conclusions and Recommendations","metadata":{}},{"cell_type":"markdown","source":"TensorFlow deep learning platform has helped us accomplish significant roles in training our dataset. We were able to build several utility layers and nodes within each layer. The number of activation layers and the type of activation node were identified to influence the performance of the model significantly. \nOne way to enhance the performance of the model is experiment with different numbers of this elements/parameters .\nOther Machine learning algorithms have employed optimization techniques like hyperparameter tuning, feature selections and enhanced feature transformations which can equally be used for CNN with tensor flow or h2o and other improved deep learning and artificial intelligence tools. \nFrom this model; 'relu' activation was found to perform better compared to the sigmoid activation. But, this does not mean that it's better than 'sigmoid'; there are variations on different activation nodes depending on the model under study.","metadata":{}},{"cell_type":"markdown","source":"## References","metadata":{}},{"cell_type":"markdown","source":"  https://www.kaggle.com/c/vinbigdata-chest-xray-abnormalities-detection/data\n  https://github.com/naitik2314/Chest-X-Ray-Medical-Diagnosis-with-Deep-Learning/blob/master/X-Ray.ipynb\n  https://www.youtube.com/watch?v=zYSB_l87ip8https://mjoc.uitm.edu.my/main/images/journal/vol4-1-2019/MJOC_vol41_seq5.pdf\n  https://stackoverflow.com/questions/55335231/using-elmo-with-tf-keras-throws-valueerror-could-not-convert-string-to-float\n  https://www.programcreek.com/python/example/110868/pydicom.read_file\n  https://www.kaggle.com/pmarcelino/comprehensive-data-exploration-with-python\n  https://www.kaggle.com/pmarcelino/comprehensive-data-exploration-with-python\n  https://www.raddq.com/dicom-processing-segmentation-visualization-in-python/\n  https://www.kaggle.com/c/data-science-bowl-2017/code\n  https://www.geeksforgeeks.org/draw-geometric-shapes-images-using-opencv/?ref=rp\n  https://pydicom.github.io/pydicom/0.9/pydicom_user_guide.html\n  https://www.kaggle.com/jiandanjinxin/first-pass-through-data-w-3d-convnet/notebook#Section-3:-Preprocessing-our-Data\n  https://www.kaggle.com/pmarcelino/comprehensive-data-exploration-with-python\n  https://www.kaggle.com/vrajparikh/initial-eda\n  https://www.w3schools.com/python/python_file_remove.asp\n  https://learning.oreilly.com/library/view/programming-computer-vision/9781449341916/ch01.html#convert_images_to_another_format\n  https://mipt-oulu.github.io/solt/DSBowl18_segmentation.html\n\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}