{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Exploratory Data Analysis As a Beginner\n\nHello there.  This is the first time I do kaggle competition, although I have been learning python for 3 years, particular in Web Scraping and Data Analysis.\n\nI geniuely present the EDA and please feel free to give comments.🙂🙂","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n#Since there are too many data, therefore I did not print it all.\n\n#File path\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        if filename.endswith('dcm') == False:\n#            print(os.path.join(dirname, filename))\n\n#/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv\n#/kaggle/input/rsna-breast-cancer-detection/train.csv\n#/kaggle/input/rsna-breast-cancer-detection/test.csv","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-30T06:28:09.083421Z","iopub.execute_input":"2022-11-30T06:28:09.083895Z","iopub.status.idle":"2022-11-30T06:28:09.090352Z","shell.execute_reply.started":"2022-11-30T06:28:09.083853Z","shell.execute_reply":"2022-11-30T06:28:09.089009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip3 install python-gdcm\n!pip3 install pylibjpeg pylibjpeg-libjpeg\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom pathlib import Path\nimport glob","metadata":{"execution":{"iopub.status.busy":"2022-11-30T05:44:06.841526Z","iopub.execute_input":"2022-11-30T05:44:06.842911Z","iopub.status.idle":"2022-11-30T05:44:36.923974Z","shell.execute_reply.started":"2022-11-30T05:44:06.842857Z","shell.execute_reply":"2022-11-30T05:44:36.922478Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntest = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-30T05:44:36.926628Z","iopub.execute_input":"2022-11-30T05:44:36.928212Z","iopub.status.idle":"2022-11-30T05:44:37.068913Z","shell.execute_reply.started":"2022-11-30T05:44:36.92814Z","shell.execute_reply":"2022-11-30T05:44:37.06757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-11-30T05:44:37.070525Z","iopub.execute_input":"2022-11-30T05:44:37.071259Z","iopub.status.idle":"2022-11-30T05:44:37.113371Z","shell.execute_reply.started":"2022-11-30T05:44:37.071219Z","shell.execute_reply":"2022-11-30T05:44:37.111907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-11-30T05:44:50.582977Z","iopub.execute_input":"2022-11-30T05:44:50.583654Z","iopub.status.idle":"2022-11-30T05:44:50.599724Z","shell.execute_reply.started":"2022-11-30T05:44:50.583583Z","shell.execute_reply":"2022-11-30T05:44:50.598839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Since the test set does not have the columns - cancer (target), biosy, invasive, BIRADS, density, difficult negative case, I believe we have to trim them down for training the model.**\n\nThough, the EDA will still rely on the unavailable data for exploring.","metadata":{}},{"cell_type":"markdown","source":"# Is there any missing data?","metadata":{}},{"cell_type":"code","source":"train_na = (train.isna().sum()).to_frame(name='Train_na')\nplt.figure(figsize=(15,4))\nsns.heatmap(data=train_na.T,cmap='Spectral').set_title('count missing values', fontsize=14)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:57:24.897785Z","iopub.execute_input":"2022-11-30T06:57:24.898226Z","iopub.status.idle":"2022-11-30T06:57:25.216802Z","shell.execute_reply.started":"2022-11-30T06:57:24.89819Z","shell.execute_reply":"2022-11-30T06:57:25.215196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since the columns will be dropped, it doesnt matter if is null.","metadata":{}},{"cell_type":"markdown","source":"# Data against the age","metadata":{}},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(11.7,8.27)})\nsns.histplot(data = train, x = 'age', hue='cancer', bins=20, kde=True, multiple='stack', palette='pastel')\nplt.title('Age Distribution for cancer patient')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Is there any missing image id for each patient?\n\nThe answer is No.  At least 4 images for each patients in the training set\n\nHowever, it is found quite a lot of patients have 4 image","metadata":{}},{"cell_type":"code","source":"sns.countplot(train.groupby('patient_id')['image_id'].count())\nplt.title(\"No. of image distribution\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:35:47.158664Z","iopub.execute_input":"2022-11-30T06:35:47.159118Z","iopub.status.idle":"2022-11-30T06:35:47.465213Z","shell.execute_reply.started":"2022-11-30T06:35:47.159064Z","shell.execute_reply":"2022-11-30T06:35:47.463477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's take a close look for patients who have more than 4 images taken**","metadata":{}},{"cell_type":"code","source":"train_4more_image = train.groupby('patient_id')['image_id'].count()[train.groupby('patient_id')['image_id'].count()>4]\nsns.countplot(train_4more_image)\nplt.title('The distribution of paitent with more than 4 images')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:35:50.647577Z","iopub.execute_input":"2022-11-30T06:35:50.64798Z","iopub.status.idle":"2022-11-30T06:35:50.880864Z","shell.execute_reply.started":"2022-11-30T06:35:50.647937Z","shell.execute_reply":"2022-11-30T06:35:50.879322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"extra_image = train[train['patient_id'].isin(train_4more_image.index.to_list())]\nextra_image","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:43:42.384355Z","iopub.execute_input":"2022-11-30T06:43:42.384824Z","iopub.status.idle":"2022-11-30T06:43:42.413692Z","shell.execute_reply.started":"2022-11-30T06:43:42.384788Z","shell.execute_reply":"2022-11-30T06:43:42.412298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=extra_image, x='site_id')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:44:28.960132Z","iopub.execute_input":"2022-11-30T06:44:28.960677Z","iopub.status.idle":"2022-11-30T06:44:29.190691Z","shell.execute_reply.started":"2022-11-30T06:44:28.960637Z","shell.execute_reply":"2022-11-30T06:44:29.189044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Mainly site 1 takes more images**","metadata":{}},{"cell_type":"code","source":"sns.countplot(data=extra_image, x='cancer')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:45:38.738679Z","iopub.execute_input":"2022-11-30T06:45:38.739074Z","iopub.status.idle":"2022-11-30T06:45:38.970224Z","shell.execute_reply.started":"2022-11-30T06:45:38.739042Z","shell.execute_reply":"2022-11-30T06:45:38.968437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train['patient_id']==65456]","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:50:51.398966Z","iopub.execute_input":"2022-11-30T06:50:51.399453Z","iopub.status.idle":"2022-11-30T06:50:51.426108Z","shell.execute_reply.started":"2022-11-30T06:50:51.399417Z","shell.execute_reply":"2022-11-30T06:50:51.424485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For validating my script, this patient has taken 10 images in one time.","metadata":{}},{"cell_type":"code","source":"sns.histplot(data = extra_image, x = 'age', bins=20, kde=True, multiple='stack', palette='pastel')\nplt.title('Age Distribution for more than 4 images')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:46:40.68613Z","iopub.execute_input":"2022-11-30T06:46:40.686563Z","iopub.status.idle":"2022-11-30T06:46:41.149042Z","shell.execute_reply.started":"2022-11-30T06:46:40.686531Z","shell.execute_reply":"2022-11-30T06:46:41.147451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is just made out of curiosity.  I assumed the extra image is taken due to blur image, which should be likely to happen when the patient is older.","metadata":{}},{"cell_type":"markdown","source":"# Data VS Site_ID","metadata":{}},{"cell_type":"code","source":"sns.countplot(data=train, x='site_id')\nplt.title('Number of Source from each hospital')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:59:34.734834Z","iopub.execute_input":"2022-11-30T06:59:34.73526Z","iopub.status.idle":"2022-11-30T06:59:34.888006Z","shell.execute_reply.started":"2022-11-30T06:59:34.735228Z","shell.execute_reply":"2022-11-30T06:59:34.887163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"implant0 = train[train['site_id']==1].groupby(['cancer'])['cancer'].count()\nimplant1 = train[train['site_id']==2].groupby(['cancer'])['cancer'].count()\ncolors = sns.color_palette('pastel')[0:5]\nfig = plt.figure(figsize=(10,10),dpi=144)\nax1 = fig.add_subplot(121)\nax1.pie(implant0.to_list(),implant0.index, autopct='%1.1f%%', colors=colors)\nax1.set_title('Cancer Rate at site 1')\nax2 = fig.add_subplot(122)\nax2.pie(implant1.to_list(),implant1.index, autopct='%1.1f%%', colors=colors)\nax2.set_title('Cancer Rate at site 2')\nax2.legend(['No Cancer','Cancer'], loc='best')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:18:09.360187Z","iopub.execute_input":"2022-11-30T06:18:09.361328Z","iopub.status.idle":"2022-11-30T06:18:09.724692Z","shell.execute_reply.started":"2022-11-30T06:18:09.361274Z","shell.execute_reply":"2022-11-30T06:18:09.723104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Seems the site id has nothing to do with the diagnose.","metadata":{}},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(11.7,8.27)})\nsns.countplot(data=train, x='implant')\nplt.title('The propotation of implantation')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:54:38.826146Z","iopub.execute_input":"2022-11-30T06:54:38.826636Z","iopub.status.idle":"2022-11-30T06:54:39.061066Z","shell.execute_reply.started":"2022-11-30T06:54:38.826599Z","shell.execute_reply":"2022-11-30T06:54:39.059581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"implant0 = train[train['implant']==0].groupby(['cancer'])['cancer'].count()\nimplant1 = train[train['implant']==1].groupby(['cancer'])['cancer'].count()\ncolors = sns.color_palette('pastel')[0:5]\nfig = plt.figure(figsize=(10,10),dpi=144)\nax1 = fig.add_subplot(121)\nax1.pie(implant0.to_list(),implant0.index, autopct='%1.1f%%', colors=colors)\nax1.set_title('Cancer Rate with no implantation')\nax2 = fig.add_subplot(122)\nax2.pie(implant1.to_list(),implant1.index, autopct='%1.1f%%', colors=colors)\nax2.set_title('Cancer Rate with implantation')\nax2.legend(['No Cancer','Cancer'], loc='best')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:55:00.175108Z","iopub.execute_input":"2022-11-30T06:55:00.175571Z","iopub.status.idle":"2022-11-30T06:55:00.540073Z","shell.execute_reply.started":"2022-11-30T06:55:00.175533Z","shell.execute_reply":"2022-11-30T06:55:00.538957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(11.7,8.27)})\nsns.countplot(data=train, x='difficult_negative_case')\nplt.title('Number of Difficulty Negative Case')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:17:04.368838Z","iopub.execute_input":"2022-11-30T06:17:04.369327Z","iopub.status.idle":"2022-11-30T06:17:04.582909Z","shell.execute_reply.started":"2022-11-30T06:17:04.36929Z","shell.execute_reply":"2022-11-30T06:17:04.58202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"diff0 = train[train['difficult_negative_case']==0].groupby(['cancer'])['cancer'].count()\ndiff1 = train[train['difficult_negative_case']==1].groupby(['cancer'])['cancer'].count()\ncolors = sns.color_palette('pastel')[0:5]\nfig = plt.figure(figsize=(10,10),dpi=144)\nax1 = fig.add_subplot(121)\nax1.pie(diff0.to_list(),diff0.index, autopct='%1.1f%%', colors=colors)\nax1.set_title('Cancer Rate with difficult_negative_case')\nax2 = fig.add_subplot(122)\nax2.pie(diff1.to_list(),diff1.index, autopct='%1.1f%%', colors=colors)\nax2.set_title('Cancer Rate with difficult_negative_case')\nax1.legend(['No Cancer','Cancer'], loc='best')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:56:20.954215Z","iopub.execute_input":"2022-11-30T06:56:20.954675Z","iopub.status.idle":"2022-11-30T06:56:21.293365Z","shell.execute_reply.started":"2022-11-30T06:56:20.954639Z","shell.execute_reply":"2022-11-30T06:56:21.29182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Yeah.  It means it is difficulty but negative.  It is very noral that i have 0% if the value is 1**","metadata":{}},{"cell_type":"code","source":"# source: https://www.kaggle.com/code/allunia/rsna-csf-cervical-spine-fracture-eda/notebook\nfrom math import ceil\n# Since it could be more than 4 images, the subplot function should be amended to fit this situation\ndef rescale_img_to_hu(dcm_ds):\n    \"\"\"Rescales the image to Hounsfield unit.\"\"\"\n    data = dcm_ds.pixel_array\n    if dcm_ds.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    return data * dcm_ds.RescaleSlope + dcm_ds.RescaleIntercept\n\ndef show_images_for_patient(patient_id):\n    patient_dir = os.path.join('../input/rsna-breast-cancer-detection/train_images', str(patient_id))\n    num_images = len(glob.glob(f\"{patient_dir}/*\"))\n    print(f\"Number of images for patient: {num_images}\")\n    num_col = ceil(num_images / 2)\n    fig, axs = plt.subplots(2,num_col, figsize=(24,15))\n    axs = axs.flatten()\n    for i, img_path in enumerate(list(Path(patient_dir).iterdir())):\n        ds = pydicom.dcmread(img_path)\n        axs[i].imshow(rescale_img_to_hu(ds), cmap=\"bone\")","metadata":{"execution":{"iopub.status.busy":"2022-11-30T07:06:45.059819Z","iopub.execute_input":"2022-11-30T07:06:45.060274Z","iopub.status.idle":"2022-11-30T07:06:45.070981Z","shell.execute_reply.started":"2022-11-30T07:06:45.060238Z","shell.execute_reply":"2022-11-30T07:06:45.069471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_images_for_patient(65456)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T07:06:50.464206Z","iopub.execute_input":"2022-11-30T07:06:50.464623Z","iopub.status.idle":"2022-11-30T07:07:17.934838Z","shell.execute_reply.started":"2022-11-30T07:06:50.464589Z","shell.execute_reply":"2022-11-30T07:07:17.933257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_images_for_patient(10175)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T07:07:17.937232Z","iopub.execute_input":"2022-11-30T07:07:17.937825Z","iopub.status.idle":"2022-11-30T07:07:30.716062Z","shell.execute_reply.started":"2022-11-30T07:07:17.937755Z","shell.execute_reply":"2022-11-30T07:07:30.714473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is my first notebook for this competition\n\nI wish for some comments on my work.  Thank you","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}