{"cells":[{"metadata":{},"cell_type":"markdown","source":"\n# Thoracic lung diseases :\n### Thoracic disorders are conditions of the heart, lungs, mediastinum, esophagus, chest wall, diaphragm and great vessels and may include:\n* Chronic obstructive pulmonary disease (COPD)\n* Pulmonary embolism\n* Lungs cancer,.....,etc\n"},{"metadata":{},"cell_type":"markdown","source":"## Importing the libraries"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"from glob import glob # to read files\nfrom os.path import splitext\nfrom random import choice\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom  matplotlib import colors\nimport seaborn as sns\nimport missingno as msno  #to visualize the missing values\nimport plotly.express as px\n\nimport pydicom\nfrom pydicom import read_file\n\nimport skimage\nfrom skimage.io import imread\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Get a deep insights for our dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/vinbigdata-chest-xray-abnormalities-detection/train.csv')\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"shape = df.shape\nprint('The shape of our datase:'+\" \"+str(shape))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### *Note:* That the size of the csv file is \"67914\" , Meanwhile in the overview it was mentined that we have 15,000 independently-labeled images and will be evaluated on a test set of 3,000 images ,So; for sure there is a duplication in our data we will handle this ."},{"metadata":{},"cell_type":"markdown","source":"# Dealing with the duplicated records\n### Firstly exploring them "},{"metadata":{"trusted":true},"cell_type":"code","source":"df['image_id'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.loc[df['image_id']=='ecf474d5d4f65d7a3e23370a68b8c6a0',:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"duplication = df['image_id'].duplicated().sum()\nprint('The count of the duplication in our dataset:'+' '+str(duplication))\nprint('Unique value : '+\" \"+str(shape[0]-duplication))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Reading the whole file of the train folder"},{"metadata":{"trusted":true},"cell_type":"code","source":"pathes = glob('../input/vinbigdata-chest-xray-abnormalities-detection/train/*')\nlen(pathes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# creatin a dicionarty of key('image_id') and value ('pathes')\npathes_dict = dict()\nkeys = [splitext(x)[0].split('/')[-1] for x in pathes]\npathes_dict = {keys[i]:pathes[i] for i in range(0,len(pathes))}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# list(pathes_dict.keys())\ndf['pathes'] = df['image_id'].map(pathes_dict)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Exploring Our Class Label"},{"metadata":{},"cell_type":"markdown","source":"* Note : Having 14 class label as it was mentioned "},{"metadata":{"trusted":true},"cell_type":"code","source":"df['class_name'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Visalizng the count of class_name"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,5))\nsns.countplot(data=df ,y='class_name')\nplt.title('Counts of the Classes',fontsize=20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"list_ = ['No finding','Aortic enlargement','Cardiomegaly'\n         ,'Pulmonary fibrosis','Pleural thickening','Lung Opacity'\n         ,'Pleural effusion','Other lesion','Nodule/Mass','Infiltration'\n         ,'ILD','Calcification','Consolidation','Atelectasis','Pneumothorax']\nfig = px.pie(df,values=df['class_name'].value_counts(),names=list_ \n       , color_discrete_sequence=px.colors.sequential.RdBu)\n\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# check for the missing value and handle it"},{"metadata":{"trusted":true},"cell_type":"code","source":"df.isna().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Visualizing the missing values"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(5,5))\nmsno.bar(df)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Visulizing the locations of the missing values\nsns.heatmap(df.isna(),cmap='Blues')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"*Note :*All the missing value of(x_min , y_min,x_max,y_max)for the No finding class"},{"metadata":{},"cell_type":"markdown","source":"### Dealing with the missing values\n#### Filling the missing value with zero"},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df.fillna(0, axis=0)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check for the missing values\ncount = df.isna().sum()\nprint('The count of the missing values :'+\"\\n\"+str(count))\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Start fun with DICOM Images"},{"metadata":{},"cell_type":"markdown","source":"## **What is DICOM?**\n##### Digital Imaging and Communications in Medicine (DICOM) is the standard for the communication and management of medical imaging information and related data.DICOM is most commonly used for storing and transmitting medical images enabling the integration of medical imaging devices such as scanners, servers, workstations, printers, network hardware, and picture archiving and communication systems (PACS) from multiple manufacturers. It has been widely adopted by hospitals and is making inroads into smaller applications like dentists' and doctors' offices. \nfor more info: [https://en.wikipedia.org/wiki/DICOM](http://)"},{"metadata":{},"cell_type":"markdown","source":"## Get a deep insights about our .dicom images\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"from pydicom import read_file\nrand_img = choice(df['pathes'])\nprint('our random data :'+' '+str(rand_img))\nimg = read_file(rand_img)\n#print the meta date for the .dicom\nprint(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from skimage.transform import resize\nimport tqdm\ndef resize_img(img):\n    rescaled_img = resize(img.pixel_array,(512,512))\n    return rescaled_img\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"*Note: The dicom images have a additive informatio that we could manipulate them later on*"},{"metadata":{"trusted":true},"cell_type":"code","source":"\n# plotting image with bounding box via matplotlib.patches \ndef create_bbox(data, img):\n    fig = plt.figure() \n    ax = fig.add_subplot(111) \n    ax.imshow(resize_img(img),cmap=plt.cm.bone)\n    color_dict = {'No finding':'w','Aortic enlargement':'xkcd:sky blue','Cardiomegaly':'xkcd:green'\n             ,'Pulmonary fibrosis':'xkcd:beige','Pleural thickening':'xkcd:purple'\n                  ,'Lung Opacity':'xkcd:red','Pleural effusion':'xkcd:yellow','Other lesion':'xkcd:orange',\n                  'Nodule/Mass':'xkcd:neon green','Infiltration':'xkcd:pale orange',\n                  'ILD':'xkcd:blue','Calcification':'xkcd:white','Consolidation':'xkcd:murky green'\n                  ,'Atelectasis':'xkcd:tomato','Pneumothorax':'xkcd:puke brown'}\n    data['colors'] = data['class_name'].map(color_dict)\n    scale = 5\n    for i in range(0,len(data)):\n        x, y  =int(data.iloc[i,4])/scale, int(data.iloc[i,5])/scale\n        width, height = int(data.iloc[i,6])/scale, int(data.iloc[i,7])/scale\n        color = data.iloc[i,9]\n        rect = patches.Rectangle((x, y),\n                                         width, height,\n                                         linewidth = 1,\n                                         edgecolor = str(color) ,\n                                         facecolor = 'none')\n        ax.add_patch(rect)\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# get the data \nselected_img = df[df['pathes']==rand_img]\ncreate_bbox(selected_img,img)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# visulizing various random images:\n\nrand_list = [choice(df['pathes'])for x in range (0,5)]\n# rand_list\nfig = plt.figure(figsize=(20,10))\n\nfor i in range(0,5):\n    img = read_file(rand_list[i])\n    bbox_info = df[df['pathes']==rand_list[i]]\n    create_bbox(bbox_info,img)\n\nfig.show()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Getting the additional infomation from .dicom in our dataframe"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}