{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Converting the first DICOM image into an array of values:"},{"metadata":{"trusted":true},"cell_type":"code","source":"import pydicom\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image = pydicom.dcmread('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/9a5094b2563a1ef3ff50dc5c7ff71345.dicom')\nimage","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image.PatientAge # Access patient metadata from the dcmread function's output","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.iloc[2]['x_min']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Adding a bounding box on above image, based on x, y coordinates in the train.csv file"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(1)\nax.imshow(image.pixel_array, cmap = \"gray\")\nlength = df.iloc[2]['x_max'] - df.iloc[2]['x_min']\nwidth = df.iloc[2]['y_max'] -  df.iloc[2]['y_min']\nbox = matplotlib.patches.Rectangle((df.iloc[2]['x_min'], df.iloc[2]['y_min']), length, width, linewidth=1, facecolor='none', edgecolor = 'r')\nax.add_patch(box)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"EDA of the classes present and understanding the distribution of classes"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Spread of images across classes\nsns.distplot(df['class_id'], kde = False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"This dataset is unbalanced and needs to be balanced for predictions."},{"metadata":{"trusted":true},"cell_type":"code","source":"len(df[df['class_id'] == 14])   # Number of rows corresponding to \"Not identified\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"67914 - 31818 # Total number of records that have a specific category (total - unidentified)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(df.dropna(subset=[\"x_min\", \"x_max\", \"y_min\", \"y_max\"]))\n# Once all Nan values for x_min, x_max, y_min, y_max are removed, we get the same number as above\n# Conclusion: If x, y coordinates are not known, then that row belongs to type 14\n# This can be checked by removing all Nan rows and checking if any are of type 14 in the pending df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ided_df = df.dropna(subset=[\"x_min\", \"x_max\", \"y_min\", \"y_max\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ided_df['class_id'].unique() # Check unique values for all remaining objects, after removing Nan\n# Here class_id 14 is not present, proving above hypothesis.\n# Thus, if x and y coordinates are missing, it cannot be identified as belonging to any class\n# The model needs to be trained only on the remaining set of values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ided_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Plotting distribution of classes in the new df\nsns.distplot(ided_df[\"class_id\"], kde = False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Classes 0, 3, 11 and 13 still have a large number of samples as compared to other classes."},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}