{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# &#128218; Imports","metadata":{}},{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-05T11:57:18.765104Z","iopub.execute_input":"2022-12-05T11:57:18.765546Z","iopub.status.idle":"2022-12-05T11:57:35.480736Z","shell.execute_reply.started":"2022-12-05T11:57:18.765509Z","shell.execute_reply":"2022-12-05T11:57:35.4792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport gdcm\nimport glob\nfrom IPython.display import display\nimport os\nimport re\n\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport pydicom\nimport seaborn as sns\nfrom tqdm.notebook import tqdm\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T11:57:35.483626Z","iopub.execute_input":"2022-12-05T11:57:35.484448Z","iopub.status.idle":"2022-12-05T11:57:36.531772Z","shell.execute_reply.started":"2022-12-05T11:57:35.484399Z","shell.execute_reply":"2022-12-05T11:57:36.530828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# &#128221; Input data","metadata":{}},{"cell_type":"code","source":"ROOT = '/kaggle/input/rsna-breast-cancer-detection'\ndf_train = pd.read_csv(ROOT + '/train.csv')\ndf_test = pd.read_csv(ROOT + '/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-05T11:57:36.533084Z","iopub.execute_input":"2022-12-05T11:57:36.533702Z","iopub.status.idle":"2022-12-05T11:57:36.662386Z","shell.execute_reply.started":"2022-12-05T11:57:36.53366Z","shell.execute_reply":"2022-12-05T11:57:36.661597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# &#128202; EDA tabular data (csv file)","metadata":{}},{"cell_type":"code","source":"print('train.csv')\ndisplay(df_train.head().style.background_gradient(cmap=\"Pastel1\"))\nprint('test.csv')\ndisplay(df_test.head().style.background_gradient(cmap=\"Pastel1\"))","metadata":{"execution":{"iopub.status.busy":"2022-12-05T11:57:44.968942Z","iopub.execute_input":"2022-12-05T11:57:44.96932Z","iopub.status.idle":"2022-12-05T11:57:45.027602Z","shell.execute_reply.started":"2022-12-05T11:57:44.969287Z","shell.execute_reply":"2022-12-05T11:57:45.0265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def base_info(df):\n    features = []\n    dtypes = []\n    unique_values = []\n    for col in df.columns:\n        features.append(col)\n        dtypes.append(df[col].dtype) \n        unique_values.append(len(df[col].unique()))\n    \n    df_ = pd.DataFrame({\n        'features': features,\n        'dtypes': dtypes,\n        'unique values':unique_values\n    })\n    display(df_.style.background_gradient(cmap=\"Pastel1\"))\n\n\ndef show_correlation(df, features):\n    df = df.loc[:, features]\n    corr = df.corr()\n    fig = plt.figure(figsize=(9, 9))\n    sns.heatmap(corr, annot=True,\n                fmt='.2f',\n                cmap=sns.color_palette('coolwarm',200))\n    plt.title('Correlation')\n    plt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-05T12:06:02.21493Z","iopub.execute_input":"2022-12-05T12:06:02.215347Z","iopub.status.idle":"2022-12-05T12:06:02.224751Z","shell.execute_reply.started":"2022-12-05T12:06:02.215313Z","shell.execute_reply":"2022-12-05T12:06:02.223608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Base info\nbase_info(df_train)\n# Correlation\nfeatures = ['site_id', 'age', 'implant', 'machine_id',\n            'biopsy', 'invasive', 'BIRADS',  \n            'difficult_negative_case', 'cancer'] \n           # remove ['patient_id', 'image_id', 'view', 'laterality', 'density']\nshow_correlation(df_train, features=features)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T12:06:02.9881Z","iopub.execute_input":"2022-12-05T12:06:02.988506Z","iopub.status.idle":"2022-12-05T12:06:03.79927Z","shell.execute_reply.started":"2022-12-05T12:06:02.988473Z","shell.execute_reply":"2022-12-05T12:06:03.798079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# &#128247; EDA images (DICOM files)","metadata":{}},{"cell_type":"markdown","source":"## Meta data\nSome meta data are different from images as below. (Photometric Interpretation, Body Part Thickness, Compression Force, and so on...)","metadata":{}},{"cell_type":"code","source":"dcm1 = pydicom.dcmread(ROOT + '/train_images/10006/1459541791.dcm')\ndcm2 = pydicom.dcmread(ROOT + '/train_images/10226/1943220805.dcm')\nprint('1459541791.dcm:\\n', dcm1, '\\n\\n')\nprint('1943220805.dcm:\\n', dcm2)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T11:59:00.443682Z","iopub.execute_input":"2022-12-05T11:59:00.444598Z","iopub.status.idle":"2022-12-05T11:59:00.767103Z","shell.execute_reply.started":"2022-12-05T11:59:00.444549Z","shell.execute_reply":"2022-12-05T11:59:00.766253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<ul>\n  <li>(0018, 11a0) Body Part Thickness</li>\n    The average thickness in mm of the body part examined when compressed, if compression has been applied during exposure.\n  <li>(0018, 11a2) Compression Force</li>\n    The compression force applied to the body part during exposure, measured in Newtons.\n  <li>(0028, 0004) Photometric Interpretation</li>\n    Specifies the intended interpretation of the pixel data. See Section C.7.6.3.1.2 for further explanation:<br>\n    MONOCHROME1 - The minimum sample value is intended to be displayed as white after any VOI gray scale transformations have been performed.<br>\n    MONOCHROME2 - The minimum sample value is intended to be displayed as black after any VOI gray scale transformations have been performed.\n  <li>(0028, 0030) Pixel Spacing</li>\n    Physical distance in the patient between the center of each pixel, specified by a numeric pair - adjacent row spacing (delimiter) adjacent column spacing, in mm. See Section 10.7.1.3 for further explanation of the value order.\n  <li>(0028, 0100) Bits Allocated</li>\n    Number of bits allocated for each pixel sample. Each sample shall have the same number of bits allocated. Bits Allocated (0028,0100) shall be either 1, or a multiple of 8. See PS3.5 for further explanation.\n  <li>(0028, 1052) Rescale Intercept and (0028, 1053) Rescale Slope</li>\n    Rescale Intercept b, Rescale Slope m in relationship between stored values (SV):<br>\n    Output units = m*SV + b.\n</ul>\nref. https://dicom.innolitics.com/ciods/digital-x-ray-image/dx-positioning/001811a2","metadata":{}},{"cell_type":"code","source":"def get_dicom_meta_data(sampling_size=50):\n    \n    sample_index = df_train.sample(sampling_size, random_state=2022).index\n    patient_ids = df_train.loc[sample_index, 'patient_id'].to_list()\n    image_ids = df_train.loc[sample_index, 'image_id'].to_list()\n    dcm_paths = [ROOT + f'/train_images/{patient_id}/{image_id}.dcm'\n                 for patient_id, image_id in zip(patient_ids, image_ids)]\n    \n    # Metadata\n    BodyPartThickness = []\n    CompressionForce = []\n    PhotometricInterpretation = []\n    PixelSpacing = []\n    BitsAllocated = []\n    RescaleIntercept = []\n    RescaleSlope = []\n    Rows = []\n    Columns = []\n    \n    for dcm_path in tqdm(dcm_paths):\n        dcm = pydicom.dcmread(dcm_path)\n        PhotometricInterpretation.append(dcm.PhotometricInterpretation)\n        try:\n            BodyPartThickness.append(dcm.BodyPartThickness)\n        except:\n             BodyPartThickness.append(None)\n        try:\n            CompressionForce.append(dcm.CompressionForce)\n        except:\n            CompressionForce.append(None)\n        try:\n            PixelSpacing.append(dcm.PixelSpacing)\n        except:\n            PixelSpacing.append(None)\n        BitsAllocated.append(dcm.BitsAllocated)    \n        RescaleIntercept.append(dcm.RescaleIntercept)\n        RescaleSlope.append(dcm.RescaleSlope)\n        Rows.append(dcm.Rows)\n        Columns.append(dcm.Columns)    \n\n    df_dcm = pd.DataFrame({\n        'patient_id':patient_ids,\n        'image_id':image_ids,\n        'PhotometricInterpretation': PhotometricInterpretation,\n        'BodyPartThickness': BodyPartThickness,\n        'CompressionForce': CompressionForce,\n        'PixelSpacing': PixelSpacing,\n        'BitsAllocated': BitsAllocated,\n        'RescaleIntercept': RescaleIntercept,\n        'RescaleSlope': RescaleSlope,\n        'Rows': Rows,\n        'Columns': Columns,\n        })\n    \n    return df_dcm\n\n\ndef plot_hist_dicom_meta(df):\n    fig, ax = plt.subplots(2, 4, figsize=(22,10))\n    ax = ax.flatten()\n    sns.set_palette(\"pastel\")\n    sns.histplot(df['PhotometricInterpretation'], ax=ax[0])\n    sns.histplot(df['BodyPartThickness'], ax=ax[1])\n    sns.histplot(df['CompressionForce'], ax=ax[2])\n    sns.countplot(df['BitsAllocated'], ax=ax[3], linewidth=1.0, edgecolor='black')\n    sns.countplot(df['RescaleIntercept'], ax=ax[4], linewidth=1.0, edgecolor='black')\n    sns.countplot(df['RescaleSlope'], ax=ax[5], linewidth=1.0, edgecolor='black')\n    sns.histplot(df['Rows'], ax=ax[6])\n    sns.histplot(df['Columns'], ax=ax[7])\n    ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-05T11:59:00.890061Z","iopub.execute_input":"2022-12-05T11:59:00.891007Z","iopub.status.idle":"2022-12-05T11:59:00.90717Z","shell.execute_reply.started":"2022-12-05T11:59:00.890968Z","shell.execute_reply":"2022-12-05T11:59:00.906017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_dcm = get_dicom_meta_data(sampling_size=5000)\ndisplay(df_dcm.head(10).style.background_gradient(cmap=\"Pastel1\"))\nplot_hist_dicom_meta(df_dcm)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T12:07:00.585585Z","iopub.execute_input":"2022-12-05T12:07:00.58609Z","iopub.status.idle":"2022-12-05T12:07:02.079378Z","shell.execute_reply.started":"2022-12-05T12:07:00.586045Z","shell.execute_reply":"2022-12-05T12:07:02.078274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Images","metadata":{}},{"cell_type":"code","source":"def processing_pixel_data(dcm, resize=512):\n    \n    img = dcm.pixel_array\n    img = img * dcm.RescaleSlope + dcm.RescaleIntercept\n    img = (img - img.min()) / (img.max() - img.min())\n    if dcm.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n    img = cv2.resize(img, (resize, resize))\n    img = (img * 255).astype(np.uint8)\n\n    return img\n\n\ndef show_dcm_images(patient_id, df):\n    \n    root = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\n    dcm_paths = glob.glob(f'{root}/{patient_id}/*.dcm')\n    num_dcm_files = len(dcm_paths)\n     \n    # Sort dcm filed by id\n    def natural_keys(text):\n        keys = []\n        for t in re.split(r'(\\d+)', text):\n            if t.isdigit():\n                t = int(t)\n            keys.append(t)\n        return keys\n    dcm_paths.sort(key=natural_keys)\n            \n    # Plot images\n    ncols = 5\n    nrows = int(np.ceil(num_dcm_files/ncols))\n    fig, axes = plt.subplots(nrows=nrows, ncols=ncols, figsize=(25,5*nrows))\n    fig.suptitle(f'patient_id: {patient_id}', size=20)\n    axes = axes.flatten()\n    for i in range(ncols*nrows):\n        \n        axes[i].axis('off')\n        \n        if i >= num_dcm_files:\n            pass\n        \n        else:\n            dcm = pydicom.dcmread(dcm_paths[i])\n            img = processing_pixel_data(dcm, resize=512)\n            img_id = int(dcm_paths[i].split('/')[-1].split('.')[0])\n            diagnosis = df[df['image_id']==img_id]['cancer'].values[0]\n            axes[i].imshow(img, cmap=\"bone\")\n            axes[i].set_title(f'{\"+\" if diagnosis==1 else \"-\"}',\n                              fontsize=20,\n                              weight='bold')\n        \n    plt.show()\n    ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-05T12:03:44.089142Z","iopub.execute_input":"2022-12-05T12:03:44.08971Z","iopub.status.idle":"2022-12-05T12:03:44.103511Z","shell.execute_reply.started":"2022-12-05T12:03:44.089665Z","shell.execute_reply":"2022-12-05T12:03:44.102464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sampling_size = 10\n\nprint('Positive cancer patients')\npositive_patient_ids = \\\n    df_train[df_train['cancer']==1]['patient_id'].unique()[:sampling_size]\nfor p_id in tqdm(positive_patient_ids):\n    show_dcm_images(p_id, df_train)\n    \nprint('Negative cancer patients')\nnegative_patient_ids = \\\n    df_train[df_train['cancer']==0]['patient_id'].unique()[:sampling_size]\nfor p_id in tqdm(negative_patient_ids):\n    show_dcm_images(p_id, df_train)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T12:03:44.104751Z","iopub.execute_input":"2022-12-05T12:03:44.105049Z","iopub.status.idle":"2022-12-05T12:04:37.849157Z","shell.execute_reply.started":"2022-12-05T12:03:44.105022Z","shell.execute_reply":"2022-12-05T12:04:37.848011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}