{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom glob import glob\n\n!pip install python-gdcm\n!pip install pylibjpeg-libjpeg\n\nimport pydicom","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-30T23:06:30.05801Z","iopub.execute_input":"2022-11-30T23:06:30.058414Z","iopub.status.idle":"2022-11-30T23:06:47.540311Z","shell.execute_reply.started":"2022-11-30T23:06:30.058387Z","shell.execute_reply":"2022-11-30T23:06:47.539092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set(style='darkgrid', font_scale=1.4)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T23:06:47.542196Z","iopub.execute_input":"2022-11-30T23:06:47.542723Z","iopub.status.idle":"2022-11-30T23:06:47.550722Z","shell.execute_reply.started":"2022-11-30T23:06:47.542694Z","shell.execute_reply":"2022-11-30T23:06:47.548412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"code","source":"#reading all dcm files into train and test\ntrain = glob(\"/kaggle/input/rsna-breast-cancer-detection/train_images/*/*.dcm\")\ntest = glob(\"/kaggle/input/rsna-breast-cancer-detection/test_images/*/*.dcm\")\nprint(\"train df files: \", len(train))\nprint(\"test df files: \", len(test))\n","metadata":{"execution":{"iopub.status.busy":"2022-11-30T23:06:47.552545Z","iopub.execute_input":"2022-11-30T23:06:47.552945Z","iopub.status.idle":"2022-11-30T23:07:00.979526Z","shell.execute_reply.started":"2022-11-30T23:06:47.55291Z","shell.execute_reply":"2022-11-30T23:07:00.978584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/rsna-breast-cancer-detection/train.csv\")\ntrain_df.head()\nprint(train_df.shape)\nprint(train_df.patient_id.nunique())","metadata":{"execution":{"iopub.status.busy":"2022-11-30T22:19:12.400471Z","iopub.execute_input":"2022-11-30T22:19:12.400902Z","iopub.status.idle":"2022-11-30T22:19:12.512956Z","shell.execute_reply.started":"2022-11-30T22:19:12.400879Z","shell.execute_reply":"2022-11-30T22:19:12.511518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see the number of images matches the amount of rows in the training set. But the number of unique patients is significantly less as patients can have multiple images in dataset.","metadata":{}},{"cell_type":"markdown","source":"## Image Viewer","metadata":{}},{"cell_type":"code","source":"# displaying the image\nsample = train[100]\nimg = pydicom.read_file(sample).pixel_array\nplt.imshow(img, cmap=plt.cm.bone)\nplt.grid(None)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-30T23:01:36.016022Z","iopub.execute_input":"2022-11-30T23:01:36.016361Z","iopub.status.idle":"2022-11-30T23:01:38.184913Z","shell.execute_reply.started":"2022-11-30T23:01:36.016337Z","shell.execute_reply":"2022-11-30T23:01:38.183537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# finding an image with a positive label\ncancer_df = train_df.loc[train_df.cancer == 1]\ncancer_im = cancer_df.iloc[[1]]\nim = f\"/kaggle/input/rsna-breast-cancer-detection/train_images/{cancer_im.patient_id.item()}/{cancer_im.image_id.item()}.dcm\"","metadata":{"execution":{"iopub.status.busy":"2022-11-30T23:01:38.187026Z","iopub.execute_input":"2022-11-30T23:01:38.187414Z","iopub.status.idle":"2022-11-30T23:01:38.198498Z","shell.execute_reply.started":"2022-11-30T23:01:38.187387Z","shell.execute_reply":"2022-11-30T23:01:38.197317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = pydicom.read_file(im).pixel_array\nplt.imshow(img, cmap=plt.cm.bone)\nplt.grid(None)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T23:01:39.031807Z","iopub.execute_input":"2022-11-30T23:01:39.032244Z","iopub.status.idle":"2022-11-30T23:01:40.606645Z","shell.execute_reply.started":"2022-11-30T23:01:39.032215Z","shell.execute_reply":"2022-11-30T23:01:40.605906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA train set","metadata":{}},{"cell_type":"markdown","source":"**Missing Values**","metadata":{}},{"cell_type":"code","source":"print('Train SET MISSING VALUES:')\nprint(train_df.isna().sum())","metadata":{"execution":{"iopub.status.busy":"2022-11-30T23:01:43.566885Z","iopub.execute_input":"2022-11-30T23:01:43.567516Z","iopub.status.idle":"2022-11-30T23:01:43.584842Z","shell.execute_reply.started":"2022-11-30T23:01:43.567488Z","shell.execute_reply":"2022-11-30T23:01:43.583497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T23:01:48.248349Z","iopub.execute_input":"2022-11-30T23:01:48.248691Z","iopub.status.idle":"2022-11-30T23:01:48.278551Z","shell.execute_reply.started":"2022-11-30T23:01:48.248666Z","shell.execute_reply":"2022-11-30T23:01:48.276298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Figure size\nplt.figure(figsize=(6,6))\n\n# Pie plot\ntrain_df['cancer'].value_counts().plot.pie(explode=[0.1,0.1], autopct='%1.1f%%', shadow=True, textprops={'fontsize':16}).set_title(\"Target distribution\")","metadata":{"execution":{"iopub.status.busy":"2022-11-30T23:01:52.743635Z","iopub.execute_input":"2022-11-30T23:01:52.744043Z","iopub.status.idle":"2022-11-30T23:01:52.878116Z","shell.execute_reply.started":"2022-11-30T23:01:52.744015Z","shell.execute_reply":"2022-11-30T23:01:52.876754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Insights: the data is massively skewed, we may want to try oversampling to even up the target labels distribution","metadata":{}},{"cell_type":"code","source":"# Expenditure features\nnum_feats=['age']\n\n# Plot expenditure features\nfig=plt.figure(figsize=(10,20))\nfor i, var_name in enumerate(num_feats):\n    # Left plot\n    ax=fig.add_subplot(2,1,2*i+1)\n    sns.histplot(data=train_df.loc[train_df.cancer == 0], x=var_name, axes=ax, bins=50, kde=True)\n    ax.set_title(\"age distribution of patients without cancer\")\n    \n    # Right plot (truncated)\n    ax=fig.add_subplot(2,1,2*i+2)\n    sns.histplot(data=train_df.loc[train_df.cancer == 1], x=var_name, axes=ax, bins=50, kde=True)\n    ax.set_title(\"age distribution of patients with cancer\")\nfig.tight_layout()  # Improves appearance a bit\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T23:01:54.026466Z","iopub.execute_input":"2022-11-30T23:01:54.026818Z","iopub.status.idle":"2022-11-30T23:01:54.867617Z","shell.execute_reply.started":"2022-11-30T23:01:54.026792Z","shell.execute_reply":"2022-11-30T23:01:54.86625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_feats=['site_id', 'laterality', 'view', 'machine_id']\n\n# Plot categorical features\nfig=plt.figure(figsize=(10,16))\nfor i, var_name in enumerate(cat_feats):\n    ax=fig.add_subplot(4,1,i+1)\n    sns.countplot(data=train_df, x=var_name, axes=ax, hue='cancer')\n    ax.set_title(var_name)\nfig.tight_layout()  # Improves appearance a bit\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T23:02:05.153584Z","iopub.execute_input":"2022-11-30T23:02:05.153925Z","iopub.status.idle":"2022-11-30T23:02:05.926317Z","shell.execute_reply.started":"2022-11-30T23:02:05.153899Z","shell.execute_reply":"2022-11-30T23:02:05.924402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2 = train_df.groupby(['patient_id'])['patient_id'].count()\nax = df2.plot.hist(bins=10)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T23:02:10.8922Z","iopub.execute_input":"2022-11-30T23:02:10.892537Z","iopub.status.idle":"2022-11-30T23:02:11.106547Z","shell.execute_reply.started":"2022-11-30T23:02:10.892512Z","shell.execute_reply":"2022-11-30T23:02:11.105169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see the majority of the patients have 4 images or records in the dataset ","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}