{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os \nfrom glob import glob\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\n\ndataset_folder = '/kaggle/input/vinbigdata-chest-xray-abnormalities-detection' \ntrain_csv = pd.read_csv(os.path.join(dataset_folder, 'train.csv'))\n\nfix, axes = plt.subplots(1, 2, figsize =(16,8)) \nplt.xticks(fontsize = 14, rotation = 90)\nplt.yticks(fontsize = 14)\n\nsns.countplot(train_csv.rad_id, ax = axes[0])\naxes[0].set_title(label='Distribution of reviewers')\n\nsns.countplot(train_csv.class_name, ax = axes[1])\naxes[1].set_title(label=\"Distribution of labels\")\n\n#print(f'min duplicates of images: {occurrences.min()}, max: {occurrences.max()}')\n#print(f'reviews by radi: {reviewes_by_rad_id}')\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class index:\n    def __init__(self, xmax, ymax):\n        self.xmax = xmax\n        self.ymax = ymax\n        self.i = 0\n        self.x = 0\n        self.y = 0\n        \n    def inc(self):\n        self.i += 1\n        if self.x != self.xmax - 1:\n            self.x+=1\n        else:\n            self.x=0\n            self.y+=1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"xmax, ymax = (6,3) \nind = index(xmax, ymax)\nfig, axes = plt.subplots(xmax, ymax, figsize=(24,24), constrained_layout=True)\nplt.xticks(fontsize = 14, rotation = 90)\nplt.yticks(fontsize = 14)\n# fig.tight_layout()\n\nfor rad_id in sorted(train_csv['rad_id'].unique()):\n    filtered_ids = train_csv.query(f'rad_id == \"{rad_id}\"')\n    sns.countplot(filtered_ids.class_name, ax = axes[ind.x, ind.y])\n    axes[ind.x, ind.y].set_title(label=f'{rad_id}')\n    axes[ind.x, ind.y].tick_params(labelrotation=90)\n    ind.inc()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}