{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n'''for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))'''\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-01T11:33:36.653348Z","iopub.execute_input":"2022-12-01T11:33:36.654057Z","iopub.status.idle":"2022-12-01T11:33:36.687933Z","shell.execute_reply.started":"2022-12-01T11:33:36.653959Z","shell.execute_reply":"2022-12-01T11:33:36.687065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\nimport plotly.graph_objects as go","metadata":{"execution":{"iopub.status.busy":"2022-12-01T12:01:48.476063Z","iopub.execute_input":"2022-12-01T12:01:48.476698Z","iopub.status.idle":"2022-12-01T12:01:48.48271Z","shell.execute_reply.started":"2022-12-01T12:01:48.47665Z","shell.execute_reply":"2022-12-01T12:01:48.481396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-01T11:34:06.314017Z","iopub.execute_input":"2022-12-01T11:34:06.314535Z","iopub.status.idle":"2022-12-01T11:34:06.44225Z","shell.execute_reply.started":"2022-12-01T11:34:06.314493Z","shell.execute_reply":"2022-12-01T11:34:06.441145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA ","metadata":{}},{"cell_type":"code","source":"train_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-12-01T11:34:39.95405Z","iopub.execute_input":"2022-12-01T11:34:39.95455Z","iopub.status.idle":"2022-12-01T11:34:39.987756Z","shell.execute_reply.started":"2022-12-01T11:34:39.954513Z","shell.execute_reply":"2022-12-01T11:34:39.986804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-01T11:34:28.562162Z","iopub.execute_input":"2022-12-01T11:34:28.562626Z","iopub.status.idle":"2022-12-01T11:34:28.574291Z","shell.execute_reply.started":"2022-12-01T11:34:28.562586Z","shell.execute_reply":"2022-12-01T11:34:28.573077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-12-01T12:32:45.883378Z","iopub.execute_input":"2022-12-01T12:32:45.883983Z","iopub.status.idle":"2022-12-01T12:32:45.915169Z","shell.execute_reply.started":"2022-12-01T12:32:45.883917Z","shell.execute_reply":"2022-12-01T12:32:45.914064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-01T12:33:49.37683Z","iopub.execute_input":"2022-12-01T12:33:49.37725Z","iopub.status.idle":"2022-12-01T12:33:49.437305Z","shell.execute_reply.started":"2022-12-01T12:33:49.377218Z","shell.execute_reply":"2022-12-01T12:33:49.43615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(train_df , \n             y=train_df.cancer.value_counts().values, \n             x=train_df.cancer.value_counts().index, \n             template=  \"simple_white\"  ,\n             title= \"Count of cancer and noncancer images\",\n             color = train_df.cancer.value_counts().values,\n             color_continuous_scale='Burg',\n              labels={'x':'Cancer or not (1 = Cancer, 0 = Noncancer)', 'y':\"Count of images\"},\n             text=['{}'.format(p) for p in train_df.cancer.value_counts().values])\n\nfig.update_layout(coloraxis_showscale=False)\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-12-01T11:57:17.042941Z","iopub.execute_input":"2022-12-01T11:57:17.043988Z","iopub.status.idle":"2022-12-01T11:57:17.111762Z","shell.execute_reply.started":"2022-12-01T11:57:17.043948Z","shell.execute_reply":"2022-12-01T11:57:17.110607Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.bar(train_df , \n             y=train_df.difficult_negative_case.value_counts().values, \n             x=train_df.difficult_negative_case.value_counts().index, \n             template=  \"simple_white\"  ,\n             title= \"Count of difficult negative case images\",\n             color = train_df.difficult_negative_case.value_counts().values,\n             color_continuous_scale='Burg',\n              labels={'x':'Difficult negative case', 'y':\"Count of images\"},\n             text=['{}'.format(p) for p in train_df.difficult_negative_case.value_counts().values])\n\nfig.update_layout(coloraxis_showscale=False)\nfig.show()\n\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-01T12:42:12.201935Z","iopub.execute_input":"2022-12-01T12:42:12.202305Z","iopub.status.idle":"2022-12-01T12:42:12.267002Z","shell.execute_reply.started":"2022-12-01T12:42:12.202273Z","shell.execute_reply":"2022-12-01T12:42:12.265905Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.box(train_df.groupby('patient_id')['age','cancer'].max() , \n             y=train_df.age,    \n             template=  \"simple_white\"  ,\n             x= train_df.cancer,\n             title= \"Distribution of patients age based on whether have cancer or not\",\n             labels={'x':'Cancer or not (1 = Cancer, 0 = Noncancer)', 'y':\"Distribution of patients age\"},\n             color_discrete_sequence=px.colors.sequential.Burg_r)\n\nfig.update_layout(coloraxis_showscale=False)\nfig.show()\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-01T11:58:06.459959Z","iopub.execute_input":"2022-12-01T11:58:06.460649Z","iopub.status.idle":"2022-12-01T11:58:06.530637Z","shell.execute_reply.started":"2022-12-01T11:58:06.460613Z","shell.execute_reply":"2022-12-01T11:58:06.529571Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_df = train_df.groupby(['cancer','laterality'])['patient_id'].size().reset_index(name=\"counts\").sort_values(by='counts')\nnew_df['Percentage'] = new_df.groupby('cancer')['counts'].apply(lambda x: 100 * (x/x.sum()))\nnew_df = new_df.pivot(index='laterality', columns='cancer')['Percentage'].fillna(0)\nnew_df = np.round(new_df, 1)\n\nfig = go.Figure(go.Heatmap(z = new_df.values, \n                           x=new_df.columns, \n                           y=new_df.index, \n                           type = 'heatmap',\n                           colorscale ='Burg', \n                           showscale = False, \n                           text=new_df.values.astype(str), \n                           textfont={\"size\":10},\n                           hovertemplate='Diagnosed with cancer: %{x}<br>٬eft or right breast: %{y}<br>Counts: %{z}<extra></extra>',\n                           texttemplate=\"%{text}%\"))\n\nfig.update_layout(coloraxis_showscale=False,\n                  title=\"The Percentage of diagnosed with cancer or not, based on the laterality of breast image\",)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-01T12:24:02.175638Z","iopub.execute_input":"2022-12-01T12:24:02.176033Z","iopub.status.idle":"2022-12-01T12:24:02.213675Z","shell.execute_reply.started":"2022-12-01T12:24:02.175998Z","shell.execute_reply":"2022-12-01T12:24:02.212417Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_machine_id = train_df.groupby(by=[\"patient_id\",'cancer','machine_id']).size().reset_index(name=\"counts\")\ndf_machine_id['machine_id'] = [('Machine-'+str(i) )for i in df_machine_id['machine_id']]\nfig = px.pie( values=df_machine_id.counts,\n             names=df_machine_id.machine_id, \n             facet_col=df_machine_id.cancer,\n             color_discrete_sequence = px.colors.sequential.Burg_r, \n             title='The distribution of machines that used to take image')\nfig.for_each_annotation(lambda a: a.update(text= \"diagnosed with cancer = \"+ a.text.split(\"=\")[1]))\nfig.update_traces(textposition='inside', textinfo='percent+label')\nfig.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-12-01T12:31:19.679681Z","iopub.execute_input":"2022-12-01T12:31:19.680027Z","iopub.status.idle":"2022-12-01T12:31:19.799966Z","shell.execute_reply.started":"2022-12-01T12:31:19.679998Z","shell.execute_reply":"2022-12-01T12:31:19.798899Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read Images","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pydicom\nds = pydicom.dcmread('/kaggle/input/rsna-breast-cancer-detection/train_images/10006/1459541791.dcm')\n\nplt.imshow(ds.pixel_array, cmap=plt.cm.bone) ","metadata":{"execution":{"iopub.status.busy":"2022-12-01T12:39:40.38336Z","iopub.execute_input":"2022-12-01T12:39:40.384171Z","iopub.status.idle":"2022-12-01T12:39:43.463635Z","shell.execute_reply.started":"2022-12-01T12:39:40.384109Z","shell.execute_reply":"2022-12-01T12:39:43.462383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}