{"cells":[{"metadata":{},"cell_type":"markdown","source":"## RSNA ICH Detection\n**Objective**: To detect the different sub-type of Intracranial Hemorrhage (ICH)  \n**Eval Metric**: weighted multi-label logarithmic loss  \n*predicted probabilities are replaced with max(min(p,1−10−15),10−15).*  \n\n![](https://www.googleapis.com/download/storage/v1/b/kaggle-user-content/o/inbox%2F603584%2F56162e47358efd77010336a373beb0d2%2Fsubtypes-of-hemorrhage.png?generation=1568657910458946&alt=media)  \n\n- There are 5 sub-types of ICH (epidural, intraparenchymal, intraventricular, subarachnoid, subdural)    \n- 'any' is a boolean of ANY sub-type of ICH\n- Severeness of 5 sub-types varies. (Can we say for sure one sub-type is more serious than the others? If so, can we use ordinal regression like in the diabetic retinopathy competition?)\n- Differences between the sub-types"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport pydicom\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"TRAIN_DIR = '../input/rsna-intracranial-hemorrhage-detection/stage_1_train_images/'\nTEST_DIR = '../input/rsna-intracranial-hemorrhage-detection/stage_1_test_images/'\n\ntrain_img = [TRAIN_DIR + f for f in os.listdir(TRAIN_DIR)]\ntest_img = [TEST_DIR + f for f in os.listdir(TEST_DIR)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.listdir('../input/rsna-intracranial-hemorrhage-detection')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_submission = pd.read_csv('../input/rsna-intracranial-hemorrhage-detection/stage_1_sample_submission.csv')\ndf_train = pd.read_csv('../input/rsna-intracranial-hemorrhage-detection/stage_1_train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Try to use pydicom to analyze the images\nfilename = train_img[0]\nds = pydicom.dcmread(filename)\nprint(ds)\nplt.imshow(ds.pixel_array, cmap=plt.cm.bone)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Clean the dataset\ndf_train['id_code'] = df_train.ID.apply(lambda x: 'ID_'+x.split('_')[-2])\ndf_train['subtype'] = df_train.ID.apply(lambda x: x.split('_')[-1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check if sub-type are mutually exclusive\ndf_train[['id_code','Label']].groupby('id_code').sum().Label.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"subtype_lst = df_train.subtype.unique()\n\ndf = pd.DataFrame()\ndf['id_code'] = df_train.id_code.unique()\nfor n in subtype_lst:\n    temp_df = df_train[df_train.subtype==n][['id_code','Label']].rename(columns={'Label': n})\n    df = df.merge(temp_df, on='id_code', how='left')\n    del temp_df","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Note: Different sub-types can happens at the same time"},{"metadata":{},"cell_type":"markdown","source":"Correct way to view DICOM CT images  \nhttps://www.kaggle.com/omission/eda-view-dicom-images-with-correct-windowing  \nby Richard McKinley"},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_first_of_dicom_field_as_int(x):\n    #get x[0] as in int is x is a 'pydicom.multival.MultiValue', otherwise get int(x)\n    if type(x) == pydicom.multival.MultiValue:\n        return int(x[0])\n    else:\n        return int(x)\n\ndef get_windowing(data):\n    dicom_fields = [data[('0028','1050')].value, #window center\n                    data[('0028','1051')].value, #window width\n                    data[('0028','1052')].value, #intercept\n                    data[('0028','1053')].value] #slope\n    return [get_first_of_dicom_field_as_int(x) for x in dicom_fields]\n\ndef dcm2image(data):\n    window_center, window_width, intercept, slope = get_windowing(data)\n    img = data.pixel_array\n    img = (img*slope +intercept)\n    img_min = window_center - window_width//2\n    img_max = window_center + window_width//2\n    img[img<img_min] = img_min\n    img[img>img_max] = img_max\n    return img ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# matplotlib.cm options: Sequential (2)\n#['binary', 'gist_yarg', 'gist_gray', 'gray', 'bone', 'pink', \n#'spring', 'summer', 'autumn', 'winter', 'cool', 'Wistia',\n#'hot', 'afmhot', 'gist_heat', 'copper']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"SUBTYPE = 'intraparenchymal'\nfig=plt.figure(figsize=(30, 20))\nfig.suptitle(SUBTYPE, fontsize=20)\ncolumns = 5; rows = 4\n#lst = df_train[(df_train.subtype==SUBTYPE)&(df_train.Label==1)].head(columns*rows).id_code.tolist()\nlst = df[df[SUBTYPE]==1].id_code.tolist()\n\nfor i in range(1, columns*rows +1):\n    ds = pydicom.dcmread(TRAIN_DIR + lst[i-1]+'.dcm')\n    fig.add_subplot(rows, columns, i)\n    \n    img = dcm2image(ds)\n    plt.imshow(img, cmap=plt.cm.bone)\n    fig.add_subplot","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"SUBTYPE = 'intraventricular'\nfig=plt.figure(figsize=(30, 20))\nfig.suptitle(SUBTYPE, fontsize=20)\ncolumns = 5; rows = 4\n#lst = df_train[(df_train.subtype==SUBTYPE)&(df_train.Label==1)].head(columns*rows).id_code.tolist()\nlst = df[df[SUBTYPE]==1].id_code.tolist()\n\nfor i in range(1, columns*rows +1):\n    ds = pydicom.dcmread(TRAIN_DIR + lst[i-1]+'.dcm')\n    fig.add_subplot(rows, columns, i)\n    \n    img = dcm2image(ds)\n    plt.imshow(img, cmap=plt.cm.bone)\n    fig.add_subplot","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"SUBTYPE = 'subarachnoid'\nfig=plt.figure(figsize=(30, 20))\nfig.suptitle(SUBTYPE, fontsize=20)\ncolumns = 5; rows = 4\n#lst = df_train[(df_train.subtype==SUBTYPE)&(df_train.Label==1)].head(columns*rows).id_code.tolist()\nlst = df[df[SUBTYPE]==1].id_code.tolist()\n\nfor i in range(1, columns*rows +1):\n    ds = pydicom.dcmread(TRAIN_DIR + lst[i-1]+'.dcm')\n    fig.add_subplot(rows, columns, i)\n    \n    img = dcm2image(ds)\n    plt.imshow(img, cmap=plt.cm.bone)\n    fig.add_subplot","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"SUBTYPE = 'subdural'\nfig=plt.figure(figsize=(30, 20))\nfig.suptitle(SUBTYPE, fontsize=20)\ncolumns = 5; rows = 4\n#lst = df_train[(df_train.subtype==SUBTYPE)&(df_train.Label==1)].head(columns*rows).id_code.tolist()\nlst = df[df[SUBTYPE]==1].id_code.tolist()\n\nfor i in range(1, columns*rows +1):\n    ds = pydicom.dcmread(TRAIN_DIR + lst[i-1]+'.dcm')\n    fig.add_subplot(rows, columns, i)\n    \n    img = dcm2image(ds)\n    plt.imshow(img, cmap=plt.cm.bone)\n    fig.add_subplot","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"SUBTYPE = 'epidural'\nfig=plt.figure(figsize=(30, 20))\nfig.suptitle(SUBTYPE, fontsize=20)\ncolumns = 5; rows = 4\n#lst = df_train[(df_train.subtype==SUBTYPE)&(df_train.Label==1)].head(columns*rows).id_code.tolist()\nlst = df[df[SUBTYPE]==1].id_code.tolist()\n\nfor i in range(1, columns*rows +1):\n    ds = pydicom.dcmread(TRAIN_DIR + lst[i-1]+'.dcm')\n    fig.add_subplot(rows, columns, i)\n    \n    img = dcm2image(ds)\n    plt.imshow(img, cmap=plt.cm.bone)\n    fig.add_subplot","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"SUBTYPE = 'any'\nfig=plt.figure(figsize=(30, 20))\nfig.suptitle('HEALTHY', fontsize=20)\ncolumns = 5; rows = 4\n#lst = df_train[(df_train.subtype==SUBTYPE)&(df_train.Label==0)].head(columns*rows).id_code.tolist()\nlst = df[df[SUBTYPE]==0].id_code.tolist()\n\nfor i in range(1, columns*rows +1):\n    ds = pydicom.dcmread(TRAIN_DIR + lst[i-1]+'.dcm')\n    fig.add_subplot(rows, columns, i)\n    \n    img = dcm2image(ds)\n    plt.imshow(img, cmap=plt.cm.bone)\n    fig.add_subplot","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SUBTYPE = 'any'\nfig=plt.figure(figsize=(30, 20))\nfig.suptitle('SUPER UNHEALTHY', fontsize=20)\ncolumns = 5; rows = 4\n\nmask = (df_train[['id_code','Label']].groupby('id_code').sum()==6).values\n#lst = df_train[['id_code','Label']].groupby('id_code').sum()[mask].reset_index().id_code.tolist()\nlst = df[df.sum(1) == 6].id_code.tolist()\n\nfor i in range(1, columns*rows +1):\n    ds = pydicom.dcmread(TRAIN_DIR + lst[i-1]+'.dcm')\n    fig.add_subplot(rows, columns, i)\n    \n    img = dcm2image(ds)\n    plt.imshow(img, cmap=plt.cm.bone)\n    fig.add_subplot","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"In the 'super unhealthy' images, the 1st, 2nd, 5th, 8th, 12th, 16th images all have the same marks.  \nAre they the same patient or is it the same device used to take the CT?"},{"metadata":{"trusted":true},"cell_type":"code","source":"SUBTYPE = 'any'\nfig=plt.figure(figsize=(30, 20))\nfig.suptitle('SAME PATIENT?', fontsize=20)\ncolumns = 3; rows = 2\n\nmask = (df_train[['id_code','Label']].groupby('id_code').sum()==6).values\nlst = df[df.sum(1) == 6].id_code.tolist()\nidx = [0, 1, 4, 7, 11,15]\nselect_lst = [lst[i] for i in idx]\n\nfor i in range(1, columns*rows +1):\n    ds = pydicom.dcmread(TRAIN_DIR + select_lst[i-1]+'.dcm')\n    fig.add_subplot(rows, columns, i)\n    \n    img = dcm2image(ds)\n    plt.imshow(img, cmap=plt.cm.bone)\n    fig.add_subplot","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Look into the details\nfor n in select_lst:\n    filename = TRAIN_DIR + select_lst[0] + '.dcm'\n    ds = pydicom.dcmread(filename)\n    print(ds.PatientID)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# These images are from the same patient with different CT slice\n# Maybe we should group these images by PatientID and split by Patient ID in train-val split\n\n# P.S. I just learnt that having different slices of the same patient is a normal pratice of CT scans,\n# so bear with me if the finding above looks stupid to you haha","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.DataFrame(df[subtype_lst[0]].value_counts()).rename(columns={'epidural':'n'})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dist = pd.DataFrame()\nfor n in subtype_lst:\n    dist[n+'_count'] = df[n].value_counts().values\n    dist[n+'_perc'] = df[n].value_counts(normalize=True).values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dist","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Distribution of Sub-types"},{"metadata":{"trusted":true},"cell_type":"code","source":"fig=plt.figure(figsize=(24, 5))\nfig.suptitle('Distribution of sub-types', fontsize=10)\ncolumns = 6; rows = 1\n\nlst = df.columns.tolist()\nlst.remove('id_code')\nfor i in range(1, columns*rows +1):\n    ax = fig.add_subplot(rows, columns, i)\n    ax.set_title(lst[i-1], fontsize=10)\n    #ax.bar([0,1], dist[lst[i-1]+'_perc'], color=['xkcd:sky blue', 'xkcd:orange'])\n    sns.barplot([0,1], dist[lst[i-1]+'_perc'])\n    fig.add_subplot","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Correlation"},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.heatmap(df.corr(method='pearson'))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Adding in PatientID"},{"metadata":{"trusted":true},"cell_type":"code","source":"#ds.PatientID\n#ds.StudyInstanceUID\n#ds.SeriesInstanceUID","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"patient_lst = []\nstudy_lst = []\nseries_lst = []\nfor n in df.id_code.values:\n    temp_ds = pydicom.dcmread(TRAIN_DIR + n + '.dcm')\n    patient_lst.append(temp_ds.PatientID)\n    study_lst.append(temp_ds.StudyInstanceUID)\n    series_lst.append(temp_ds.SeriesInstanceUID)\n\ndf['PatientID'] = patient_lst\ndf['StudyInstanceUID'] = study_lst\ndf['SeriesInstanceUID'] = series_lst","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.to_csv('train_df_with_UID.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}