{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Table of Contents:\n* [Univariate Frequencies](#1)\n* [Check structure for an invididual Image](#2)\n* [Bivariate Perspective (Classes/Radiologists)](#3)\n* [Pivot Tables for Classes and Radiologists](#4)\n* [Geometry](#5)"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# packages\n\n# standard\nimport numpy as np\nimport pandas as pd\n\n# plots\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# import training data file\ndf = pd.read_csv('../input/vinbigdata-chest-xray-abnormalities-detection/train.csv')\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# dimensions\ndf.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id='1'></a>\n# Univariate Frequencies"},{"metadata":{"trusted":true},"cell_type":"code","source":"# image id\ndf.image_id.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# image id - plot frequencies (top 25)\nfig = plt.figure(figsize = (12,4))\ndf.image_id.value_counts()[0:25].plot(kind='bar', color='darkred')\nplt.title('25 images with most rows in train.csv')\nplt.ylabel('Number of rows in train.csv')\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### We have 67'914 rows of data, but only 15'000 unique images. Each image occurs at least 3 times in train.csv."},{"metadata":{},"cell_type":"markdown","source":"### Classes"},{"metadata":{"trusted":true},"cell_type":"code","source":"# classes (name)\nprint(df.class_name.value_counts())\ndf.class_name.value_counts().plot(kind='bar')\nplt.title('Class Name')\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Radiologists"},{"metadata":{"trusted":true},"cell_type":"code","source":"# radiologists\nprint(df.rad_id.value_counts())\ndf.rad_id.value_counts().plot(kind='bar', color='darkgreen')\nplt.title('Radiologists')\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### R8, R9, R10 did much more labelling than the other radiologists."},{"metadata":{},"cell_type":"markdown","source":"<a id='2'></a>\n# Check structure for an individual image"},{"metadata":{"trusted":true},"cell_type":"code","source":"# let's pick an image with lots of labels\ndf_example = df[df.image_id=='fa109c087e46fe1ea27e48ce6d154d2f']\ndf_example","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# classes for this image\ndf_example.class_name.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# radiologists for this image\ndf_example.rad_id.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# cross table\npd.crosstab(df_example.class_name, df_example.rad_id)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id='3'></a>\n# Bivariate perspective"},{"metadata":{"trusted":true},"cell_type":"code","source":"# cross table of classes / radiologists\nctab = pd.crosstab(df.class_name, df.rad_id)\nctab","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# visualize as heatmap\nfig = plt.figure(figsize = (18,10))\nsns.heatmap(ctab, annot=True)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### R1, R3, R4, R5, R6 and R7 have no findings at all. R2 has only 3 findings."},{"metadata":{},"cell_type":"markdown","source":"# > Normalize table for each radiologist (column)"},{"metadata":{"trusted":true},"cell_type":"code","source":"# normalize each column\nctab_norm_rad = ctab / ctab.sum()\n# and visualize result as heatmap\nfig = plt.figure(figsize = (18,10))\nsns.heatmap(ctab_norm_rad, annot=True)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### R11, R12, R13, R14, R15, R16 and R17 have mostly (>= 80%) no findings.\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"# let's plot the individual distributions again as bar charts\nmy_list = ['R11','R12','R13','R14','R15','R16','R17']\nfor rad in my_list:\n    ctab_norm_rad[rad].plot(kind='bar')\n    plt.title(rad)\n    plt.grid()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Only R8, R9 and R10 show a somewhat diversified labelling."},{"metadata":{"trusted":true},"cell_type":"code","source":"# let's plot the individual distributions again as bar charts\nmy_list = ['R8','R9','R10']\nfor rad in my_list:\n    ctab_norm_rad[rad].plot(kind='bar')\n    plt.title(rad)\n    plt.grid()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# > Normalize table for each class (row)"},{"metadata":{"trusted":true},"cell_type":"code","source":"# normalize each row\nctab_norm_class = (ctab.transpose() / ctab.sum(axis=1)).transpose()\n# and visualize result as heatmap\nfig = plt.figure(figsize = (18,10))\nsns.heatmap(ctab_norm_class, annot=True)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# example: pick the row of \"No finding\"\nnofind_dist_on_rad = ctab_norm_class[ctab_norm_class.index=='No finding']\n\nfig = plt.figure(figsize = (12,4))\nplt.bar(x=nofind_dist_on_rad.columns.to_list(),\n        height=np.asarray(nofind_dist_on_rad).flatten(),\n        color='darkgreen')\nplt.title('Distribution of no findings across radiologists (sum=100%)')\nplt.grid()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id='4'></a>\n# Pivot Tables for classes and radiologists"},{"metadata":{},"cell_type":"markdown","source":"#### We want to reduce the training data to 15'000 rows corresponding to the 15'000 distinct images. For this we aggregate the labels and store the counts in new columns. Of course, we are losing information by doing this!"},{"metadata":{"trusted":true},"cell_type":"code","source":"df_pivot_class = pd.pivot_table(data=df[['image_id','class_name','class_id']],\n                                # class_id is used as dummy column for counting only\n                                index='image_id', # group by image_id\n                                columns=['class_name'], # new columns created from classes\n                                fill_value=0,\n                                aggfunc='count' # count values\n                               )\ndf_pivot_class = df_pivot_class.class_id\n# add count of labels\ndf_pivot_class['sum_labels'] = df_pivot_class.sum(axis=1)\n\n# preview\ndf_pivot_class.head(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### We can use this to find out the images w/o relevant labels:"},{"metadata":{"trusted":true},"cell_type":"code","source":"df_empty = df_pivot_class[df_pivot_class['No finding'] == df_pivot_class.sum_labels]\ndf_empty","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Conclusion: 10'606 of 15'000 images (70.7%) have no (non-trivial) labels!"},{"metadata":{"trusted":true},"cell_type":"code","source":"# visualize (subset of) pivot table\nn_sub = 50\nplt.figure(figsize=(8,12))\nsns.heatmap(df_pivot_class.iloc[0:n_sub,0:15])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Pivot Table for radiologists:"},{"metadata":{"trusted":true},"cell_type":"code","source":"df_pivot_rad = pd.pivot_table(data=df[['image_id','rad_id','class_id']],\n                              # class_id is used as dummy column for counting only\n                              index='image_id', # group by image_id\n                              columns=['rad_id'], # new columns created from rad_id's\n                              fill_value=0,\n                              aggfunc='count' # count values\n                             )\ndf_pivot_rad = df_pivot_rad.class_id\n\n# add distinct count of radiologists and count of labels\ntemp1 = np.count_nonzero(df_pivot_rad, axis=1)\ntemp2 = df_pivot_rad.sum(axis=1)\ndf_pivot_rad['n_rads'] = temp1\ndf_pivot_rad['sum_labels'] = temp2\n\n# preview\ndf_pivot_rad.head(10)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### We always have exactly 3 radiologist looking at an image (see also data description):"},{"metadata":{"trusted":true},"cell_type":"code","source":"df_pivot_rad.n_rads.value_counts() # check counts","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# visualize (subset of) pivot table\nn_sub = 50\nplt.figure(figsize=(8,12))\nsns.heatmap(df_pivot_rad.iloc[0:n_sub,0:17])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# make pivot tables available for download\ndf_pivot_class.to_csv('df_pivot_class.csv')\ndf_pivot_rad.to_csv('df_pivot_rad.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<a id='5'></a>\n# Geometry"},{"metadata":{"trusted":true},"cell_type":"code","source":"# use only rows having findings\ndfxy = df[df.class_name != 'No finding'].copy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# define some new features\ndfxy['dx'] = dfxy.x_max - dfxy.x_min   # width\ndfxy['dy'] = dfxy.y_max - dfxy.y_min   # height\ndfxy['dxdy'] = dfxy.dx * dfxy.dy       # pixel area\ndfxy['dy_over_dx'] = dfxy.dy / dfxy.dx # aspect ratio\n\nfeatures_num = ['x_min', 'x_max', 'y_min', 'y_max', \n                'dx', 'dy', 'dxdy', 'dy_over_dx']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# basic stats\ndfxy[features_num].describe()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Plot coordinate based features"},{"metadata":{"trusted":true},"cell_type":"code","source":"for f in features_num:\n    plt.figure(figsize=(10,4))\n    dfxy[f].plot(kind='hist', bins=200)\n    plt.grid()\n    plt.title(f)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Scatter Plots"},{"metadata":{},"cell_type":"markdown","source":"### Straight pairwise scatter plot"},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.pairplot(data=dfxy[features_num],\n             plot_kws={'alpha': 0.1})\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Now colored by class"},{"metadata":{"trusted":true},"cell_type":"code","source":"# use only the main features\nsns.pairplot(dfxy[['class_name','x_min','x_max','y_min','y_max']], \n             hue='class_name',\n             plot_kws={'alpha': 0.5})\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Nice from an aesthetic perspective, but probably too much information. Let's reduce the plot to only two classes:"},{"metadata":{"trusted":true},"cell_type":"code","source":"dfxy_sub = dfxy[dfxy.class_name.isin(['Cardiomegaly','Aortic enlargement'])]\n\nsns.pairplot(dfxy_sub[['class_name','rad_id','x_min','x_max','y_min','y_max']], \n             hue='class_name',\n             plot_kws={'alpha': 0.5})\nplt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}