{"cells":[{"metadata":{},"cell_type":"markdown","source":"### Visualization of the dataset distribution.\n1. Libraries \n2. Directories \n3. Dataframe of the training annotations\n4. Visualization of the distribution\n\n### 1. Libraries \n"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport tensorflow as tf\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 2. Directory of the training annotations CSV.  "},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"topDir = '/kaggle/'\nos.chdir(topDir)\nAnnoCSVDir = os.path.join(topDir,'input/vinbigdata-chest-xray-abnormalities-detection/train.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 3. Mapping the training CSV into dataframe and extracting important variables"},{"metadata":{"trusted":true},"cell_type":"code","source":"Training_Annotation = pd.read_csv(AnnoCSVDir)\nvalue = Training_Annotation['class_name'].value_counts()\nind = Training_Annotation['class_name'].value_counts().index\n\npercents = value/value.sum()*100","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 4. Visualization of the distribution"},{"metadata":{"trusted":true},"cell_type":"code","source":"bar_width = 0.9 # width of the bar\nopacity = 0.5  # Opacity of the fill of the bar\n\nfig = plt.figure(figsize=(15,15)) # figure size\nax = fig.add_subplot(111)\nax.barh(ind, value, bar_width, alpha=opacity,color='b') # horizontal bar plot\nax.set_yticklabels(ind, size = 12);\n\nplt.xlabel('Counts', fontsize=14);\nplt.ylabel('Thoracic Abnormality Type',fontsize=14);\nplt.title('Thoracic Abnormalities and their Distribution \\nin the Training Dataset', fontsize=14);\n\nfor i, y in enumerate(ax.patches):\n    label_per = percents[i]\n    ax.text(y.get_width()+.09, y.get_y()+.4, str(round((y.get_width()), 1)), fontsize=12)\n    ax.text(y.get_width()+.09, y.get_y()+.1, str(f'{round((label_per), 2)}%'), fontsize=12, color='m')\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}