{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Detect Hemorrhage Visualization"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n%matplotlib inline\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ncolor = sns.color_palette()\nimport os\nfrom skimage import exposure\nimport pydicom\nimport glob\nfrom pydicom.data import get_testdata_files","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/rsna-intracranial-hemorrhage-detection/stage_1_train.csv\")\nsample_sub = pd.read_csv(\"../input/rsna-intracranial-hemorrhage-detection/stage_1_sample_submission.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_path = (\"/kaggle/input/rsna-intracranial-hemorrhage-detection/stage_1_train_images\")\ntest_path = (\"/kaggle/input/rsna-intracranial-hemorrhage-detection/stage_1_test_images\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Train shape : {}\".format(train.shape))\nprint(\"Test shape : {}\".format(sample_sub.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['ImageID'] = train['ID'].apply(lambda x: 'ID_' + x.split('_')[1] + '.dcm')\ntrain['type'] = train['ID'].apply(lambda x: x.split('_')[2])\nsample_sub ['ImageID'] =sample_sub['ID'].apply(lambda x: 'ID_' + x.split('_')[1] + '.dcm')\nsample_sub['type'] = sample_sub['ID'].apply(lambda x: x.split('_')[2])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('There are {} images in the train data set'.format(len(train.ImageID.unique())))\nprint('There are {} images in the test data set'.format(len(sample_sub.ImageID.unique())))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.type.unique().tolist()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"So we need to predict the probabilty of each type of Hemorrhage is present in every picture.\n\nLet's see the distribution of label first"},{"metadata":{"trusted":true},"cell_type":"code","source":"value_dict = train.Label.value_counts().to_dict()\nprint(train.Label.value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots()\nsns.barplot(x=list(value_dict.keys()), y=list(value_dict.values()), ax=ax)\nax.set_title(\"the number of labels\")\nax.set_xlabel(\"class\")\nprint('{:.1f} % of images have at least one type of Hemorrhage.'.format((value_dict[1]/value_dict[0])*100))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"type_dict = train[train['Label'] == 1].type.value_counts().to_dict()\nprint(train[train['Label'] == 1].type.value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(figsize = (10, 6))\nsns.barplot(x=list(type_dict.keys()), y=list(type_dict.values()), ax=ax)\nax.set_title(\"the number of different type of Hemorrhage\")\nax.set_xlabel(\"type\")\ntype_dict","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images1_count = train[(train['Label'] == 1) & (train['type'] != 'any')].pivot_table(values = 'Label', index = ['ImageID'], aggfunc = 'sum').Label.value_counts().to_dict()\n# exclude the type \"any\"\nfig, ax = plt.subplots(figsize = (10, 6))\nsns.barplot(x=list(images1_count.keys()), y=list(images1_count.values()), ax=ax)\nax.set_title(\"the number of images with different different class of Hemorrhage\")\nax.set_xlabel(\"number of Hemorrhage\")\nfor index,count in images1_count.items():\n    print('There are {} images have {} Hemorrhage'.format(count, index))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Now Let's look at the pictures"},{"metadata":{},"cell_type":"markdown","source":"### DICOM Images\nAll provided images are in DICOM format. DICOM images contain associated metadata. This will include PatientID, StudyInstanceUID, SeriesInstanceUID, and other features. You will notice some PatientIDs represented in both the stage 1 train and test sets. This is known and intentional. However, there will be no crossover of PatientIDs into stage 2 test. Additionally, per the rules, \"Submission predictions must be based entirely on the pixel data in the provided datasets.\" Therefore, you should not expect to use or gain advantage by use of this crossover in stage 1."},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"filename = pydicom.read_file(os.path.join(train_path, \"ID_000039fa0.dcm\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"filename","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"info_dict = {'SOP Instance UID': (0x08, 0x18), \n             'Modality' : (0x08, 0x60),\n             'Patient ID' : (0x10, 0x20),\n             'Study Instance UID' : (0x20, 0x0d),\n             'Series Instance UID' : (0x20, 0x0e),\n             'Study ID' : (0x20, 0x10),\n             'Image Position (Patient)' : (0x20, 0x32),\n             'Image Orientation (Patient)' : (0x20, 0x37),\n             'Samples per Pixel' : (0x28, 0x02),\n             'Photometric Interpretation': (0x28, 0x04),\n             'Rows': (0x28, 0x10),\n             'Columns': (0x28, 0x11),\n             'Pixel Spacing': (0x28, 0x30),\n             'Bits Allocated' : (0x28, 0x100),\n             'Bits Stored' : (0x28, 0x101),\n             'High Bit' : (0x28, 0x102),\n             'Pixel Representation' : (0x28, 0x103),\n             'Window Center' : (0x28, 0x1050),\n             'Window Width' : (0x28, 0x1051),\n             'Rescale Intercept' : (0x28, 0x1052),\n             'Rescale Slope' : (0x28, 0x1053)}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pivot = train[train['type'] != 'any'].pivot_table(values = 'Label', index = ['ImageID'], aggfunc = 'sum').reset_index()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's visualize pictures with different numbers of Hemorrhage"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(24, 16))\nfor j in range(4):\n    for i, image in enumerate(pivot[pivot.Label == 1].iloc[j*4:(j+1)*4,].ImageID.tolist()):\n        ax = fig.add_subplot(4, 4, j * 4 + i + 1, xticks=[], yticks=[])\n        img = np.array(pydicom.read_file(os.path.join(train_path, image)).pixel_array)\n        img = exposure.equalize_hist(img)\n        plt.imshow(img, cmap = plt.cm.bone)\n        ax.set_title('One Hemorrhage ' + image)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(24, 16))\nfor j in range(4):\n    for i, image in enumerate(pivot[pivot.Label == 2].iloc[j*4:(j+1)*4,].ImageID.tolist()):\n        ax = fig.add_subplot(4, 4, j * 4 + i + 1, xticks=[], yticks=[])\n        img = np.array(pydicom.read_file(os.path.join(train_path, image)).pixel_array)\n        img = exposure.equalize_hist(img)\n        plt.imshow(img, cmap = plt.cm.bone)\n        ax.set_title('Two Hemorrhage ' + image)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(24, 16))\nfor j in range(4):\n    for i, image in enumerate(pivot[pivot.Label == 3].iloc[j*4:(j+1)*4,].ImageID.tolist()):\n        ax = fig.add_subplot(4, 4, j * 4 + i + 1, xticks=[], yticks=[])\n        img = np.array(pydicom.read_file(os.path.join(train_path, image)).pixel_array)\n        img = exposure.equalize_hist(img)\n        plt.imshow(img, cmap = plt.cm.bone)\n        ax.set_title('Three Hemorrhage ' + image)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(24, 16))\nfor j in range(4):\n    for i, image in enumerate(pivot[pivot.Label == 4].iloc[j*4:(j+1)*4,].ImageID.tolist()):\n        ax = fig.add_subplot(4, 4, j * 4 + i + 1, xticks=[], yticks=[])\n        img = np.array(pydicom.read_file(os.path.join(train_path, image)).pixel_array)\n        img = exposure.equalize_hist(img)\n        plt.imshow(img, cmap = plt.cm.bone)\n        ax.set_title('Four Hemorrhage ' + image)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(24, 16))\nfor j in range(4):\n    for i, image in enumerate(pivot[pivot.Label == 5].iloc[j*4:(j+1)*4,].ImageID.tolist()):\n        ax = fig.add_subplot(4, 4, j * 4 + i + 1, xticks=[], yticks=[])\n        img = np.array(pydicom.read_file(os.path.join(train_path, image)).pixel_array)\n        img = exposure.equalize_hist(img)\n        plt.imshow(img, cmap = plt.cm.bone)\n        ax.set_title('Five Hemorrhage ' + image)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Extract the information in DICOM image"},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_info(data, prefix) : \n    for keys, values in info_dict.items():\n        data[keys] = data['ImageID'].apply(lambda x: (pydicom.read_file(os.path.join(prefix, x))[values].value))\n    return data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = get_info(train, train_path)\nsample_sub = get_info(sample_sub, test_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.to_pickle('train_info.pkl')\nsample_sub.to_pickle('test_info.pkl')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# means = train.groupby('type').mean()\n# sample_sub.loc[sample_sub['type'] == 'epidural', 'Label'] = means.Label[1]\n# sample_sub.loc[sample_sub['type'] == 'intraparenchymal', 'Label'] = means.Label[2]\n# sample_sub.loc[sample_sub['type'] == 'intraventricular', 'Label'] = means.Label[3]\n# sample_sub.loc[sample_sub['type'] == 'subarachnoid', 'Label'] = means.Label[4]\n# sample_sub.loc[sample_sub['type'] == 'subdural', 'Label'] = means.Label[5]\n# sample_sub.loc[sample_sub['type'] == 'any', 'Label'] = means.Label[0]\n# sample_sub[['ID', 'Label']].to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}