{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries and Dataset","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cufflinks as cf\nimport os\nfrom pathlib import Path\n\nimport random\n\nimport glob as gb  \nimport pydicom as dicom\nfrom pydicom import dcmread\n\nimport plotly.express as px\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nfrom plotly.subplots import make_subplots\n\nfrom IPython.display import display_html","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:33:58.14119Z","iopub.execute_input":"2022-12-10T20:33:58.141663Z","iopub.status.idle":"2022-12-10T20:33:58.149283Z","shell.execute_reply.started":"2022-12-10T20:33:58.141625Z","shell.execute_reply":"2022-12-10T20:33:58.147915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntest = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-12-10T20:33:58.366125Z","iopub.execute_input":"2022-12-10T20:33:58.367187Z","iopub.status.idle":"2022-12-10T20:33:58.447796Z","shell.execute_reply.started":"2022-12-10T20:33:58.367137Z","shell.execute_reply":"2022-12-10T20:33:58.446752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Analysis","metadata":{}},{"cell_type":"code","source":"print('Train Data')\ndisplay(train.head())\nprint('Test Data')\ntest.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:33:58.594149Z","iopub.execute_input":"2022-12-10T20:33:58.59457Z","iopub.status.idle":"2022-12-10T20:33:58.624502Z","shell.execute_reply.started":"2022-12-10T20:33:58.594526Z","shell.execute_reply":"2022-12-10T20:33:58.623498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"\\033[1mNumber of records in the dataset is : {train.shape[0]}, and number of Unique Records : {train.patient_id.nunique()}\\033[0m\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:33:58.831235Z","iopub.execute_input":"2022-12-10T20:33:58.831905Z","iopub.status.idle":"2022-12-10T20:33:58.841683Z","shell.execute_reply.started":"2022-12-10T20:33:58.831869Z","shell.execute_reply":"2022-12-10T20:33:58.839434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum().to_frame(name='Number of Null Values').style.background_gradient(cmap='afmhot_r')","metadata":{"execution":{"iopub.status.busy":"2022-12-10T20:33:59.011995Z","iopub.execute_input":"2022-12-10T20:33:59.012487Z","iopub.status.idle":"2022-12-10T20:33:59.038211Z","shell.execute_reply.started":"2022-12-10T20:33:59.012421Z","shell.execute_reply":"2022-12-10T20:33:59.036992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-12-10T20:33:59.203209Z","iopub.execute_input":"2022-12-10T20:33:59.203606Z","iopub.status.idle":"2022-12-10T20:33:59.211608Z","shell.execute_reply.started":"2022-12-10T20:33:59.203571Z","shell.execute_reply":"2022-12-10T20:33:59.210611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"traning data have BIRADS, density have more than 45% null values, test data there is no null value","metadata":{}},{"cell_type":"markdown","source":"# Age Analysis","metadata":{}},{"cell_type":"code","source":"age_df = train[train.age.isnull() == False].groupby('patient_id')['age'].apply(lambda x: (np.unique(x)[0]))\n\nplt.figure(figsize=(20,8))\nax = sns.histplot(age_df, bins=60,kde=True,color='purple') \nax.grid(axis='y', linestyle='-', alpha=0.4) \n\nsns.set_style(\"whitegrid\")\nax.text(x=0.9, y=0.97, transform=ax.transAxes, s=\"Skewness : %f\" % age_df.skew(),fontsize=10, color='blue', fontweight='bold', verticalalignment='top', horizontalalignment='right')\nax.text(x=0.9, y=0.91, transform=ax.transAxes, s=\"Kurtosis : %f\" % age_df.kurt(),fontsize=10, color='blue', fontweight='bold', verticalalignment='top', horizontalalignment='right')\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / age_df.shape[0]:.2f}%\\n'\n    plt.annotate(percentage, (p.get_x() + p.get_width() / 2,p.get_height()), ha='center', va='center')\n    ax.set_title(\"\\nAge Wise Distribution of Patients\", fontsize=15, fontweight='bold')\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:33:59.754838Z","iopub.execute_input":"2022-12-10T20:33:59.755248Z","iopub.status.idle":"2022-12-10T20:34:01.011954Z","shell.execute_reply.started":"2022-12-10T20:33:59.755211Z","shell.execute_reply":"2022-12-10T20:34:01.010853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.box(age_df)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:01.013494Z","iopub.execute_input":"2022-12-10T20:34:01.014306Z","iopub.status.idle":"2022-12-10T20:34:01.120512Z","shell.execute_reply.started":"2022-12-10T20:34:01.014269Z","shell.execute_reply":"2022-12-10T20:34:01.119337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Age is having a positive skew but of only 0.103443 and kurtosis as -0.354158 \n* Age is having median value 59, and there is a peek in age between 67-69.","metadata":{}},{"cell_type":"markdown","source":"# Cancer Status Analysis","metadata":{}},{"cell_type":"code","source":"def highlightcol(val):\n    if val == 1:\n        color = 'red'\n    elif val == 0:\n        color = 'green'\n    else:\n        color = 'orange'\n    return 'color: %s' % color","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:01.121887Z","iopub.execute_input":"2022-12-10T20:34:01.122213Z","iopub.status.idle":"2022-12-10T20:34:01.128662Z","shell.execute_reply.started":"2022-12-10T20:34:01.122183Z","shell.execute_reply":"2022-12-10T20:34:01.127374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"non_unique = train.groupby('patient_id').cancer.transform('nunique').ne(1)\ncancer_notunique = train[non_unique]\n#print(cancer_notunique.groupby('patient_id').cancer.nunique().unique())\n\ndisplay_html(f\"<h3><br/>There are {cancer_notunique.patient_id.nunique()} paitents, who have cancer values True as well as False \", raw=True)\ndisplay_html(f\"<h4><br/>Sample data\", raw= True)\ndisplay(cancer_notunique[cancer_notunique['patient_id']==cancer_notunique.patient_id.unique()[0]].style.applymap(highlightcol,subset = pd.IndexSlice[:, ['cancer']]))\ndisplay_html(f\"<h3><br/>As there  is no time data available, analyising this 480 data samples with image id, considering ascending order being maintained for the paitent image id\", raw= True)\n\ncompletely_cured=0;still_have_cancer=0;cured_list=[];still_have_cancer_list=[];\n\nfor i in range(len(cancer_notunique.patient_id.unique())):\n    if i == len(cancer_notunique.patient_id.unique()):\n        count+=1\n    list_0 = [];list_1 = [];       \n    temp = cancer_notunique[cancer_notunique['patient_id']==cancer_notunique.patient_id.unique()[i]]\n    list_0 = temp[temp['cancer']==0]['image_id']\n    list_1 = temp[temp['cancer']==1]['image_id']\n        \n    if(all(i < max(list_0) for i in list_1)):\n        completely_cured +=1;cured_list.append(max(list_0))\n        \n    else:\n        still_have_cancer+=1;still_have_cancer_list.append(max(list_1))\n        \n\ndisplay_html(f\"<h3><br/>Out of this 480 samples, {completely_cured}, are completely cured, and {still_have_cancer}, still have cancer\", raw= True)\ntemp_cured = train[train['image_id'].isin(cured_list)]\nfig = go.Figure(data=[go.Table(header=dict(values=temp_cured.columns,fill_color='light gray'),\n                 cells=dict(values=[temp_cured],fill_color='lavenderblush'))])\nfig = go.Figure(data=[go.Table(header=dict(values=['site_id', 'patient_id', 'image_id','cancer'],fill_color='hotpink'),\n                 cells=dict(values=[temp_cured[temp_cured.columns[0]],temp_cured[temp_cured.columns[1]],temp_cured[temp_cured.columns[2]],temp_cured[temp_cured.columns[6]]],fill_color='lightgreen'))])\n\nfig.update_layout(title=\"Completely cured from 480 samples with latest image_id and cancer status\")\n                  \nfig.show()\ntemp_uncured = train[train['image_id'].isin(still_have_cancer_list)]\nfig = go.Figure(data=[go.Table(header=dict(values=temp_uncured.columns,fill_color='light gray'),\n                 cells=dict(values=[temp_cured],fill_color='lavenderblush'))])\nfig = go.Figure(data=[go.Table(header=dict(values=['site_id', 'patient_id', 'image_id','cancer'],fill_color='pink'),\n                 cells=dict(values=[temp_uncured[temp_uncured.columns[0]],temp_uncured[temp_uncured.columns[1]],temp_uncured[temp_uncured.columns[2]],temp_uncured[temp_uncured.columns[6]]],fill_color='skyblue'))])\n\nfig.update_layout(title=\"From 480 samples, data of patients who have not cured with latest image_id and cancer status\")\n                  \nfig.show()\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:01.131349Z","iopub.execute_input":"2022-12-10T20:34:01.131737Z","iopub.status.idle":"2022-12-10T20:34:02.062012Z","shell.execute_reply.started":"2022-12-10T20:34:01.131704Z","shell.execute_reply":"2022-12-10T20:34:02.06093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### ","metadata":{}},{"cell_type":"code","source":"cancer_unique_df =train[~non_unique]\nprint(cancer_unique_df.groupby('patient_id').cancer.nunique().unique())\ncancer_unique = cancer_unique_df [['patient_id','cancer']]\ncancer_unique.drop_duplicates(keep='first',inplace=True)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:02.063279Z","iopub.execute_input":"2022-12-10T20:34:02.063823Z","iopub.status.idle":"2022-12-10T20:34:02.089121Z","shell.execute_reply.started":"2022-12-10T20:34:02.063785Z","shell.execute_reply":"2022-12-10T20:34:02.087795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter_0_1_0_1=0;counter_0_0_1_1=0;c_0 =0;c_1=0;list_c_0=[];list_c_1=[];list_counter_0_1_0_1=[];list_counter_0_0_1_1=[];count=0\n\ntemp_df = train[train['patient_id'].isin(temp_uncured['patient_id'])][['patient_id','image_id','cancer']]## Creating dataset with uncured patients from 480 samples\n\n\nfor i in range(len(temp_df.patient_id.unique())):\n    list_image = temp_df[temp_df['patient_id']==temp_df.patient_id.unique()[i]]['image_id']\n    list_0 = temp_df[(temp_df['cancer']==0)&(temp_df['patient_id']==temp_df.patient_id.unique()[i])]['image_id']\n    list_1 = temp_df[(temp_df['cancer']==1)&(temp_df['patient_id']==temp_df.patient_id.unique()[i])]['image_id']\n    \n    c = temp_df[temp_df['image_id']==min(list_image)]['cancer'] # c will have cancer boolen for minimum of image_id for each patient_id\n    \n    if(c.values[0]==0): # patient with c as False\n        c_0+=1;list_c_0.append(temp_df.patient_id.unique()[i]) # Patients with starting 0 then 1 then 0 then 1\n        if max(list_0)>min(list_1):\n            counter_0_1_0_1+=1;list_counter_0_1_0_1.append(temp_df.patient_id.unique()[i])\n        elif max(list_0)<min(list_1):\n            counter_0_0_1_1+=1; list_counter_0_0_1_1.append(temp_df.patient_id.unique()[i])\n        else:\n            count+=1;\n            \n    else:\n        c_1+=1;list_c_1.append(temp_df.patient_id.unique()[i]) # Provide patients having starting 1 and in between 0 and finally then 1\n        \n    ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:02.091243Z","iopub.execute_input":"2022-12-10T20:34:02.091706Z","iopub.status.idle":"2022-12-10T20:34:02.708013Z","shell.execute_reply.started":"2022-12-10T20:34:02.09165Z","shell.execute_reply":"2022-12-10T20:34:02.706844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancer_df = pd.DataFrame()\n#cancer_df['Type Of Patients'] = ['Cancer Stats False from Beginning','Cancer True Status Not Changed', 'Completely Cured','0_1_0_1','0_0_1_1','1_0_1']\ncancer_df['Patients Cancer Status'] = ['Cancer Stats False from Beginning','Cancer Stats True from Beginning', 'Completely Cured','Cancer Stats False Turned True then False, Finally True','Cancer Stats Truned True','Cancer Stats True Turned False, Finally True']\ncancer_df['Count']=[cancer_unique.cancer.value_counts().values[0],cancer_unique.cancer.value_counts().values[1],completely_cured,counter_0_1_0_1,counter_0_0_1_1,c_1]\ncancer_df.set_index(cancer_df.columns[0], inplace=True)\n\ncancer_status_not_Changed = ((11427 + 6)/cancer_df['Count'].sum()*100)\n\ncancer_status_changed  = ((235+90+66+89) /cancer_df['Count'].sum()*100)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:02.709799Z","iopub.execute_input":"2022-12-10T20:34:02.710245Z","iopub.status.idle":"2022-12-10T20:34:02.722708Z","shell.execute_reply.started":"2022-12-10T20:34:02.710199Z","shell.execute_reply":"2022-12-10T20:34:02.721491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(20,8))\nfig.text(0.1, 0.95, \"Patient Cancer analysis \", fontsize=15, fontweight='bold')  \n\nsns.countplot(x = \"cancer\",data=train, palette=\"gnuplot2_r\", ax = ax[0])\nax[1] = cancer_df['Count'].plot(kind = \"bar\",color = 'purple')\n    \nfor i, ax in enumerate(ax.flatten()):\n    ax.grid(axis='y', linestyle='-', alpha=0.4) \n    \n    if i==0:t = train.shape[0];ax.set_title('Cancer Patient Status')\n    else:\n        t = cancer_df['Count'].sum();ax.set_ylim(0, cancer_df['Count'].max()+1500);ax.set_title('Swtich of Patient Cancer Status (if any)')\n        ax.text(x=0.98, y=0.97, transform=ax.transAxes, s=\"Cancer Status Not Changed from beginning --> %f\" % cancer_status_not_Changed, fontsize=10, color='blue', fontweight='bold', verticalalignment='top', horizontalalignment='right')\n        ax.text(x=0.75, y=0.93, transform=ax.transAxes, s=\"Cancer Status Changed  --> %f\" % cancer_status_changed , fontsize=10, color='black',   fontweight='bold', verticalalignment='top', horizontalalignment='right')\n        \n        \n    for p in ax.patches:\n        percentage = f'{100 * p.get_height() / t:.2f}%\\n'\n        ax.annotate(percentage, (p.get_x() + p.get_width() / 2,p.get_height()), ha='center', va='center')\n        \n\n    \nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:02.727048Z","iopub.execute_input":"2022-12-10T20:34:02.727369Z","iopub.status.idle":"2022-12-10T20:34:03.136067Z","shell.execute_reply.started":"2022-12-10T20:34:02.72734Z","shell.execute_reply":"2022-12-10T20:34:03.134917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The dataset has 95.92% of patients with cancer status as False; therefore, it is highly imbalanced\n\n* 0.05% of patients have cancer status as True, which remains the same.\n\n* 1.97%  of patients have cured of cancer.\n\n* 0.55% of patients have cancer status turned to True from False.\n\n* 0.75% of patients who have status as True turned to False, then finally to True.\n\n* 0.76% of patients who had the status as False turned to True, then to False, and finally to True.\n","metadata":{}},{"cell_type":"markdown","source":"## Number of Image taken wrt Patient","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nplt.figure(figsize=(15,5))\n\nax = sns.countplot(train.groupby('patient_id').size())\nt=train['patient_id'].nunique();ax.set_title('Image count wrt Patients', fontweight='bold');ax.set_ylim(0,9000) \n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / t:.3f}%\\n';x = p.get_x() + p.get_width() / 2 ; y = p.get_height();\n    ax.annotate(percentage, (x, y), ha='center', va='center');ax.grid(axis='y', linestyle='-', alpha=0.4) \n    \nax.text(x=0.75, y=0.77, transform=ax.transAxes, s=\"Number Patients with 4 Images --> 69.109%\", fontsize=10, color='green', fontweight='bold', verticalalignment='top', horizontalalignment='right')\nax.text(x=0.75, y=0.70, transform=ax.transAxes, s=\"Number Patients with more than 4 Images--> 30.891%\", fontsize=10, color='red',   fontweight='bold', verticalalignment='top', horizontalalignment='right')\n\nplt.show()\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:03.137947Z","iopub.execute_input":"2022-12-10T20:34:03.138509Z","iopub.status.idle":"2022-12-10T20:34:03.433173Z","shell.execute_reply.started":"2022-12-10T20:34:03.138448Z","shell.execute_reply":"2022-12-10T20:34:03.431955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* For 69.109% of patients, the number of images taken is 4 and for rest 30.891% the number of image taken is more than 4. \n\n* **Let us check if  number of images taken have any correlation with** :\n    * The  patient cancer status\n    * The site of visit of patients","metadata":{}},{"cell_type":"markdown","source":"## Image Analysis wrt to Patient Cancer Status","metadata":{}},{"cell_type":"markdown","source":"cured_list -> completely cured data\nlist_c_0.append(temp_df.patient_id.unique()[i]) - > Patients with starting 0 then 1 then 0 then 1\nlist_counter_0_1_0_1.append(temp_df.patient_id.unique()[i]) - > 0 1 0 1\nlist_counter_0_0_1_1.append(temp_df.patient_id.unique()[i]) -> 0 0 1 1\nlist_c_1.append(temp_df.patient_id.unique()[i]) - > Provide patients having starting 1 and in between 0 and finally then 1","metadata":{"_kg_hide-input":true}},{"cell_type":"code","source":"cancer_false_df = train[train['patient_id'].isin(cancer_unique[cancer_unique['cancer']==0]['patient_id'].unique())]\ncancer_true_df = train[train['patient_id'].isin(cancer_unique[cancer_unique['cancer']==1]['patient_id'].unique())]\ncompletely_cured_df = train[train['patient_id'].isin(train[train['image_id'].isin(cured_list)]['patient_id'].unique())]\ncancer_0_1_0_1_df = train[train['patient_id'].isin(list_counter_0_1_0_1)]\ncancer_0_0_1_1_df = train[train['patient_id'].isin(list_counter_0_0_1_1)]\ncancer_1_0_1_df = train[train['patient_id'].isin(list_c_1)]","metadata":{"execution":{"iopub.status.busy":"2022-12-10T20:34:03.434996Z","iopub.execute_input":"2022-12-10T20:34:03.437042Z","iopub.status.idle":"2022-12-10T20:34:03.462958Z","shell.execute_reply.started":"2022-12-10T20:34:03.436993Z","shell.execute_reply":"2022-12-10T20:34:03.461819Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_html(\"<h3><br/>To analyze the  patient cancer status wrt images taken, below details are also used\",raw=True)\ncancer_df['Normalize'] = cancer_df['Count']/cancer_df['Count'].sum()*100\ncancer_df.style.background_gradient(cmap='viridis_r')","metadata":{"execution":{"iopub.status.busy":"2022-12-10T20:34:03.465291Z","iopub.execute_input":"2022-12-10T20:34:03.465675Z","iopub.status.idle":"2022-12-10T20:34:03.482842Z","shell.execute_reply.started":"2022-12-10T20:34:03.465641Z","shell.execute_reply":"2022-12-10T20:34:03.481641Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(3, 2, figsize=(20,20))\nfig.text(0.1, 0.95, \"Analysis of cancer status or change in status wrt Image count\", fontsize=15, fontweight='bold')  \n\nsns.countplot(cancer_false_df.groupby('patient_id').size(),ax=ax[0,0])\nsns.countplot(cancer_true_df.groupby('patient_id').size(),ax=ax[0,1])\nsns.countplot(completely_cured_df.groupby('patient_id').size(),ax=ax[1,0])\nsns.countplot(cancer_0_1_0_1_df.groupby('patient_id').size(),ax=ax[1,1])\nsns.countplot(cancer_0_0_1_1_df.groupby('patient_id').size(),ax=ax[2,0])\nsns.countplot(cancer_1_0_1_df.groupby('patient_id').size(),ax=ax[2,1])\n\nt = train.patient_id.nunique()\n    \nfor i, ax in enumerate(ax.flatten()):\n    ax.grid(axis='y', linestyle='-', alpha=0.4) \n    \n    if i==0:   ax.set_ylim(0, cancer_false_df.patient_id.nunique())    ;ax.set_title('Cancer Status False From Beginning', fontweight='bold', color = 'green');ax.text(x=0.75, y=0.97, transform=ax.transAxes, s=\"Percentage Share of This Data %f\" % cancer_df['Normalize'].values[0], fontsize=10, verticalalignment='top', horizontalalignment='right')\n    elif i==1: ax.set_ylim(0, cancer_true_df.patient_id.nunique()+5)   ;ax.set_title('Cancer Status True From Beginning', fontweight='bold', color = 'red');ax.text(x=0.75, y=0.97, transform=ax.transAxes, s=\"Percentage Share of This Data %f\" % cancer_df['Normalize'].values[1], fontsize=10, verticalalignment='top', horizontalalignment='right')\n    elif i==2: ax.set_ylim(0, completely_cured_df.patient_id.nunique());ax.set_title('Completely Cured',fontweight='bold', color = 'forestgreen');ax.text(x=0.75, y=0.97, transform=ax.transAxes, s=\"Percentage Share of This Data %f\" % cancer_df['Normalize'].values[2], fontsize=10, verticalalignment='top', horizontalalignment='right')        \n    elif i==3: ax.set_ylim(0, cancer_0_1_0_1_df.patient_id.nunique())  ;ax.set_title('Initial Cancer Free Patient, whose status became True, then False and Finally True',fontweight='bold', color = 'red');ax.text(x=0.75, y=0.97, transform=ax.transAxes, s=\"Percentage Share of This Data %f\" % cancer_df['Normalize'].values[3], fontsize=10, verticalalignment='top', horizontalalignment='right')        \n    elif i==4: ax.set_ylim(0, cancer_0_0_1_1_df.patient_id.nunique())  ;ax.set_title('Initial Cancer Free Patient, whose status became True',fontweight='bold', color = 'red');ax.text(x=0.75, y=0.97, transform=ax.transAxes, s=\"Percentage Share of This Data %f\" % cancer_df['Normalize'].values[4], fontsize=10, verticalalignment='top', horizontalalignment='right')        \n    elif i==5: ax.set_ylim(0, cancer_1_0_1_df.patient_id.nunique())    ;ax.set_title('Initial Cancer True Patient, whose status became False and Finally True',fontweight='bold', color = 'red');ax.text(x=0.75, y=0.97, transform=ax.transAxes, s=\"Percentage Share of This Data %f\" % cancer_df['Normalize'].values[5], fontsize=10, verticalalignment='top', horizontalalignment='right')        \n\n        \n    for p in ax.patches:\n        percentage = f'{100 * p.get_height() / t:.2f}%\\n'\n        ax.annotate(percentage, (p.get_x() + p.get_width() / 2,p.get_height()), ha='center', va='center')\n        \n\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-10T20:34:03.484549Z","iopub.execute_input":"2022-12-10T20:34:03.485243Z","iopub.status.idle":"2022-12-10T20:34:05.050854Z","shell.execute_reply.started":"2022-12-10T20:34:03.485201Z","shell.execute_reply":"2022-12-10T20:34:05.049648Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 11 or more images are taken for cancer Free Patients, Is it due to routine checkup ?\n\n* Only 4 Images are taken for patients whose cancer status is True from beginning, and which remains the same.\n\n* 94.38% of completely cured patients only 4 images are taken, and there is decrease in number of patients with increase in photos taken. Is it due to status change of cancer detection ?\n\n* only for 10% of Patients whose Inital Status was False and Turned to True 5 images where taken and for rest only 4 images where taken.\n\n* More than 4 images are taken for 43.33% of Patients whose cancer status was initial True, then became False and Then True and Patients whose cancer status was False, then True, then False and finally True have more than. May be number of images are increased as the status changed more than 2 times.\n","metadata":{}},{"cell_type":"markdown","source":"# Site Analysis and Image Analysis wrt Sites","metadata":{}},{"cell_type":"code","source":"site_1 = train[train['site_id']==1];site_2 = train[train['site_id']==2]\n\nl1 = list(site_1.patient_id.unique());l2 = list(site_2.patient_id.unique())\n\nprint(site_1[site_1.patient_id.isin(l2)])\nprint(site_2[site_2.patient_id.isin(l1)])\n\nfinal_site_1 = site_1.drop_duplicates(keep='first',subset=\"patient_id\")\nfinal_site_2 = site_2.drop_duplicates(keep='first',subset=\"patient_id\")\n\ndisplay_html(f\"<h3><br/>Number of Patients Visiting site_1 is {round(final_site_1.shape[0]/train.patient_id.nunique()*100,2)}%, and visiting site_2 is {round(final_site_2.shape[0]/train.patient_id.nunique()*100,2)}%\", raw = True)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:05.052105Z","iopub.execute_input":"2022-12-10T20:34:05.052429Z","iopub.status.idle":"2022-12-10T20:34:05.087026Z","shell.execute_reply.started":"2022-12-10T20:34:05.0524Z","shell.execute_reply":"2022-12-10T20:34:05.085502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Image Dataset with more than 4 image count \ndf = train.groupby('patient_id').size().to_frame(name='count')\nimage_df = train[train['patient_id'].isin(df[df['count']>4].index.to_list())]\nimage_df_drop_dup= image_df.drop_duplicates(keep='first',subset=\"patient_id\")\npatient_more_4Img_list = list(image_df_drop_dup['patient_id'].values)","metadata":{"execution":{"iopub.status.busy":"2022-12-10T20:34:05.090202Z","iopub.execute_input":"2022-12-10T20:34:05.090579Z","iopub.status.idle":"2022-12-10T20:34:05.111809Z","shell.execute_reply.started":"2022-12-10T20:34:05.090545Z","shell.execute_reply":"2022-12-10T20:34:05.110721Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(20,8))\n#fig.text(0.1, 0.95, \"Site analysis \", fontsize=15, fontweight='bold')  \n\nsns.countplot(data=image_df, x='site_id',palette=\"winter_r\", ax=ax[0])\nsns.countplot(data=image_df_drop_dup, x='site_id',palette=\"gnuplot\", ax=ax[1])\n\n    \nfor i, ax in enumerate(ax.flatten()):\n    ax.grid(axis='y', linestyle='-', alpha=0.4) \n    \n    if i==0:t = image_df.shape[0];ax.set_title('Images count >4 wrt to site', fontweight='bold');ax.set_ylim(0,image_df.site_id.value_counts().values[0]+700) \n    else:t = train.patient_id.nunique();ax.set_ylim(0, image_df_drop_dup.site_id.value_counts().values[0]+700);ax.set_title('Patient with more than 4 images wrt Site [considering overall patient count]',fontweight='bold')\n        \n        \n        \n    for p in ax.patches:\n        percentage = f'{100 * p.get_height() / t:.2f}%\\n'\n        ax.annotate(percentage, (p.get_x() + p.get_width() / 2,p.get_height()), ha='center', va='center')\n           \nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:05.116801Z","iopub.execute_input":"2022-12-10T20:34:05.117149Z","iopub.status.idle":"2022-12-10T20:34:05.442925Z","shell.execute_reply.started":"2022-12-10T20:34:05.117117Z","shell.execute_reply":"2022-12-10T20:34:05.441612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* No patient is visiting both sites\n\n* Number of Patients Visiting site_1 is 48.84%, and visiting site_2 is 51.16%\n\n* For more than 4 images :\n\n   * 1. There is total of  71.199% more images taken at site_1 than at site_2\n   \n   * 2. Out of overall Patients, for 25.99% more than 4 images are taken from site 1\n   \n   * 3. Out of overall Patients, for 4.9% more than 4 images are taken from site 2\n     \n   * 4. While there are only 2.319% more patients visiting site_1, there is a 21.0899% increase in number of images taken at site_1 wrt to patients.[considering overall patient count]\n","metadata":{}},{"cell_type":"code","source":"false_temp = cancer_false_df[cancer_false_df.patient_id.isin(patient_more_4Img_list)].groupby('patient_id').size().reset_index(name='image_counts')\ns1 = cancer_false_df[['patient_id','site_id']].drop_duplicates(keep='first',subset=\"patient_id\");false_temp = false_temp.merge(s1, how ='inner', on='patient_id')\nfalse_temp['cancer_status']='false'\n\n\n#true_temp = cancer_true_df[cancer_true_df.patient_id.isin(patient_more_4Img_list)].groupby('patient_id').size().reset_index(name='image_counts')\n#s2 = cancer_true_df[['patient_id','site_id']].drop_duplicates(keep='first',subset=\"patient_id\");true_temp.merge(s2, how ='inner', on='patient_id')\n#true_temp['cancer_status']= 'true'\n\ncured_temp = completely_cured_df[completely_cured_df.patient_id.isin(patient_more_4Img_list)].groupby('patient_id').size().reset_index(name='image_counts')\ns3 = completely_cured_df[['patient_id','site_id']].drop_duplicates(keep='first',subset=\"patient_id\");cured_temp = cured_temp.merge(s3, how ='inner', on='patient_id')\ncured_temp['cancer_status']= 'completely_cured'\n\ncancer_0_1_0_1_temp = cancer_0_1_0_1_df[cancer_0_1_0_1_df.patient_id.isin(patient_more_4Img_list)].groupby('patient_id').size().reset_index(name='image_counts')\ns4 = cancer_0_1_0_1_df[['patient_id','site_id']].drop_duplicates(keep='first',subset=\"patient_id\");cancer_0_1_0_1_temp = cancer_0_1_0_1_temp.merge(s4, how ='inner', on='patient_id')\ncancer_0_1_0_1_temp['cancer_status'] = 'false_tured_true_false_true'\n\n\ncancer_0_0_1_1_temp = cancer_0_0_1_1_df[cancer_0_0_1_1_df.patient_id.isin(patient_more_4Img_list)].groupby('patient_id').size().reset_index(name='image_counts')\ns5 = cancer_0_0_1_1_df[['patient_id','site_id']].drop_duplicates(keep='first',subset=\"patient_id\");cancer_0_0_1_1_temp = cancer_0_0_1_1_temp.merge(s5, how ='inner', on='patient_id')\ncancer_0_0_1_1_temp['cancer_status'] = 'false_turned_true'\n\n\ncancer_1_0_1_temp = cancer_1_0_1_df[cancer_1_0_1_df.patient_id.isin(patient_more_4Img_list)].groupby('patient_id').size().reset_index(name='image_counts')\ns6 = cancer_1_0_1_df[['patient_id','site_id']].drop_duplicates(keep='first',subset=\"patient_id\");cancer_1_0_1_temp = cancer_1_0_1_temp.merge(s6, how ='inner', on='patient_id')\ncancer_1_0_1_temp['cancer_status']= 'true_turned_false_true'\n\n\nimage_count_gr4_df = pd.concat([false_temp,cured_temp,cancer_0_1_0_1_temp,cancer_0_0_1_1_temp,cancer_1_0_1_temp]);image_count_gr4_df['site_id'] = image_count_gr4_df['site_id'].astype(int)\nimage_count_gr4_df","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:05.445169Z","iopub.execute_input":"2022-12-10T20:34:05.445664Z","iopub.status.idle":"2022-12-10T20:34:05.517198Z","shell.execute_reply.started":"2022-12-10T20:34:05.445617Z","shell.execute_reply":"2022-12-10T20:34:05.515846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = image_count_gr4_df[image_count_gr4_df['cancer_status']!='false']\ndf_pivot = pd.pivot_table(temp, index='image_counts', columns='cancer_status', values='site_id', aggfunc='sum')\n\nfig, ax = plt.subplots(3, 1, figsize=(20,20))\n\nsns.countplot(x='image_counts', hue = 'site_id', data=image_count_gr4_df,palette=\"Purples_r\", ax =ax[0]);ax[0].set_title('Individual Images count > 4 wrt to site', fontweight='bold')\nsns.countplot(x='image_counts', hue = 'cancer_status', data=image_count_gr4_df,palette=\"flare_r\", ax =ax[1]);ax[1].set_title('Patient with more than 4 images wrt Site [considering overall patient count]',fontweight='bold')\ndf_pivot.plot.bar(stacked=True, ax =ax[2]);ax[2].set_title('Stacked Image Count for Patients who Cancer Status has changed from Initial One',fontweight='bold')\n    \nfor i, ax in enumerate(ax.flatten()):\n    ax.grid(axis='y', linestyle='-', alpha=0.4) \n    ax.legend(loc='upper right')\n                  \n    if i!=2:\n        for p in ax.patches:\n            percentage = f'{100 * p.get_height() / t:.2f}%\\n'\n            ax.annotate(percentage, (p.get_x() + p.get_width() / 2,p.get_height()), ha='center', va='center')\n    \nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:05.51924Z","iopub.execute_input":"2022-12-10T20:34:05.519722Z","iopub.status.idle":"2022-12-10T20:34:06.677288Z","shell.execute_reply.started":"2022-12-10T20:34:05.519677Z","shell.execute_reply":"2022-12-10T20:34:06.676147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# laterality ","metadata":{}},{"cell_type":"markdown","source":"- laterality - Whether the image is of the left or right breast.\n\n- view - The orientation of the image. The default for a screening exam is to capture two views per breast.\n","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(2,1, figsize=(20,10)) \n\nsns.countplot(data=train, x='laterality',palette= 'Blues_r', ax=ax[0])\nsns.countplot(data=train, x='view', hue='laterality',palette=\"Blues_r\", ax=ax[1])\nt = train.shape[0]\n    \nfor i, ax in enumerate(ax.flatten()):\n    ax.grid(axis='y', linestyle='-', alpha=0.4) \n    \n    if i==0:ax.set_title('laterality Visualization', fontweight='bold');ax.set_ylim(0,train.laterality.value_counts().values[0]+1500) \n    else:ax.set_ylim(0, train.laterality.value_counts().values[0]/2+1500);ax.set_title('View with laterality',fontweight='bold')\n        \n        \n        \n    for p in ax.patches:\n        percentage = f'{100 * p.get_height() / t:.2f}%\\n'\n        ax.annotate(percentage, (p.get_x() + p.get_width() / 2,p.get_height()), ha='center', va='center')\n           \nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:06.679214Z","iopub.execute_input":"2022-12-10T20:34:06.67988Z","iopub.status.idle":"2022-12-10T20:34:07.223981Z","shell.execute_reply.started":"2022-12-10T20:34:06.679847Z","shell.execute_reply":"2022-12-10T20:34:07.222877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* When we check laterality for overall data, there is no much difference. But we have around 18.19% of odd image count odd. Will have to check if there is any relation between these two ?\n\n* As per data description, view provide the orientation of the image. The default for a screening exam is to capture two views per breast.\n\n    * Here we have 51.01% of MLO and 48.93%of CC and it is almost same for left and right laterality\n    \n    *  We have ML(medio-lateral view) - 0.01%, latero-medial view - 0.02%, LMO (lateral-medial oblique) -0.0018% \n    \n        * [Document ](https://www.ncbi.nlm.nih.gov/pmc/articles/PMC6113143/#:~:text=In%20screening%20digital%20mammography%2C%20each,taken%20from%20above%20the%20breast.) states In screening digital mammography, each breast is typically imaged with two different views, i.e., the mediolateral oblique (MLO) view and cranial caudal (CC) view . The MLO view is taken from the center of the chest outward, while the CC view is taken from above the breast. --> Which is in par with the visualization.\n        \n    *  We have ML(medio-lateral view) - 0.01%, latero-medial view - 0.02%, LMO (lateral-medial oblique) -0.0018% \n       \n      \n\nLet us analyze laterality and CC and ML view with respect to image count\n","metadata":{}},{"cell_type":"code","source":"for i in range(len(df['count'].value_counts().index)):\n    temp_img_df  = train[train['patient_id'].isin(df[df['count']==df['count'].value_counts().index[i]].index.to_list())]\n    temp_img_view = temp_img_df[temp_img_df['view'].isin(['MLO','CC'])]\n    \n    fig, ax = plt.subplots(1,2, figsize=(20,5))\n    sns.countplot(data=temp_img_df, x='laterality',palette= 'Greens_r', ax=ax[0])\n    sns.countplot(data=temp_img_view, x='view', hue='laterality',palette=\"Greens_r\", ax=ax[1])  \n\n    t = train.shape[0];\n   \n    for j, ax in enumerate(ax.flatten()):\n        if i%2 ==0: title = 'Laterality Visualization of Image Count ' + str(df['count'].value_counts().index[i])\n        else: title = 'MLO and CC - View Visualization of Image Count ' + str(df['count'].value_counts().index[i])\n            \n        ax.set_title(title, fontweight='bold')\n        for p in ax.patches:\n            percentage = f'{100 * p.get_height() / t:.5f}%\\n'\n            ax.annotate(percentage, (p.get_x() + p.get_width() / 2,p.get_height()), ha='center', va='center')\n           \n        \n        \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-10T20:34:07.225387Z","iopub.execute_input":"2022-12-10T20:34:07.225764Z","iopub.status.idle":"2022-12-10T20:34:11.302514Z","shell.execute_reply.started":"2022-12-10T20:34:07.225731Z","shell.execute_reply":"2022-12-10T20:34:11.301217Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# biopsy","metadata":{}},{"cell_type":"markdown","source":"biopsy indicates whether or not a follow-up biopsy was performed on the breast.","metadata":{}},{"cell_type":"code","source":"train.groupby('cancer').biopsy.value_counts(normalize=True).mul(100).round(2).reset_index(name='biopsy_done_percentage')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:11.306155Z","iopub.execute_input":"2022-12-10T20:34:11.306708Z","iopub.status.idle":"2022-12-10T20:34:11.323996Z","shell.execute_reply.started":"2022-12-10T20:34:11.306673Z","shell.execute_reply":"2022-12-10T20:34:11.322681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Patient who have cancer status True, had done with 100% biopsy.\n\n* Only 3.38% of patients who have latest status have False is done with biopsy. \n\nTo understand this 3.38% of data, will have to analyze ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(3, 2, figsize=(20,20))\nfig.text(0.1, 0.95, \"Biopsy Analysis with cancer status or change in status of Paitent\", fontsize=15, fontweight='bold')  \n\nsns.countplot(data = cancer_false_df, x ='biopsy',hue='cancer', palette= 'Dark2_r',ax=ax[0,0])\nsns.countplot(data = cancer_true_df, x ='biopsy',hue='cancer',palette= 'Dark2_r',ax=ax[0,1])\nsns.countplot(data = completely_cured_df, x ='biopsy',hue='cancer',palette= 'Dark2_r',ax=ax[1,0])\nsns.countplot(data = cancer_0_1_0_1_df, x ='biopsy',hue='cancer',palette= 'Dark2_r',ax=ax[1,1])\nsns.countplot(data = cancer_0_0_1_1_df, x ='biopsy',hue='cancer',palette= 'Dark2_r',ax=ax[2,0])\nsns.countplot(data = cancer_1_0_1_df, x = 'biopsy',hue='cancer',palette= 'Dark2_r',ax=ax[2,1])\n\n    \nfor i, ax in enumerate(ax.flatten()):\n    ax.grid(axis='y', linestyle='-', alpha=0.4) \n    if   i==0: ax.set_title('Cancer Status False From Beginning', fontweight='bold', color = 'green'); t=cancer_false_df.shape[0]\n    elif i==1: ax.set_title('Cancer Status True From Beginning', fontweight='bold', color = 'red');t = cancer_true_df.shape[0]\n    elif i==2: ax.set_title('Completely Cured',fontweight='bold', color = 'forestgreen');t = completely_cured_df.shape[0]\n    elif i==3: ax.set_title('Initial Cancer Free Patient, whose status became True, then False and Finally True',fontweight='bold', color = 'red');t = cancer_0_1_0_1_df.shape[0]\n    elif i==4: ax.set_title('Initial Cancer Free Patient, whose status became True',fontweight='bold', color = 'red');t = cancer_0_0_1_1_df.shape[0]\n    elif i==5: ax.set_title('Initial Cancer True Patient, whose status became False and Finally True',fontweight='bold', color = 'red');t = cancer_1_0_1_df.shape[0]      \n        \n    for p in ax.patches:\n        percentage = f'{100 * p.get_height() / t:.2f}%\\n'\n        ax.annotate(percentage, (p.get_x() + p.get_width() / 2,p.get_height()), ha='center', va='center')\n        \n\n    \nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:11.325804Z","iopub.execute_input":"2022-12-10T20:34:11.326177Z","iopub.status.idle":"2022-12-10T20:34:12.367978Z","shell.execute_reply.started":"2022-12-10T20:34:11.326144Z","shell.execute_reply":"2022-12-10T20:34:12.367144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 2.31% of Patients whose Cancer Status was initially False, which turn became True, then False and Finally True,**biopsy was taken** when there Cancer status was False\n\n* 1.191% of Patients whose Cancer Status was initially True,  which status became False and Finally True , **biopsy was taken** when there Cancer status was False\n\n* None of Patients whose initial status was False and become True biopsy was taken in the cancer False status","metadata":{}},{"cell_type":"markdown","source":"# invasive  ","metadata":{}},{"cell_type":"markdown","source":"* invasive - If the breast is positive for cancer, whether or not the cancer proved to be invasive. Only provided for train.\n\n","metadata":{}},{"cell_type":"code","source":"cancer_true = train[train.cancer==1]\nplt.figure(figsize=(15,5))\n\nax = sns.countplot(data = cancer_true, x = 'invasive', palette='hot_r')\nt=cancer_true['patient_id'].shape[0];ax.set_title('Invasive for Cancer True Patients', fontweight='bold')\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / t:.3f}%\\n';x = p.get_x() + p.get_width() / 2 ; y = p.get_height();\n    ax.annotate(percentage, (x, y), ha='center', va='center');ax.grid(axis='y', linestyle='-', alpha=0.4) \n    \nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:12.369117Z","iopub.execute_input":"2022-12-10T20:34:12.36966Z","iopub.status.idle":"2022-12-10T20:34:12.574125Z","shell.execute_reply.started":"2022-12-10T20:34:12.36962Z","shell.execute_reply":"2022-12-10T20:34:12.572925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 70.639% of cancer True Patients, have invasive as 1, which is a serious concern.","metadata":{}},{"cell_type":"markdown","source":"# BIRADS","metadata":{}},{"cell_type":"markdown","source":"* BIRADS - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,5))\ntemp = train.copy();temp.BIRADS.fillna('Null', inplace=True)\n\nax = sns.countplot(data = temp, x = 'BIRADS', hue='cancer', palette='prism')\nt=train.shape[0];ax.set_title('BIRADS Visualization', fontweight='bold')\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / t:.3f}%\\n';x = p.get_x() + p.get_width() / 2 ; y = p.get_height();\n    ax.annotate(percentage, (x, y), ha='center', va='center');ax.grid(axis='y', linestyle='-', alpha=0.4) \n    \nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:12.57528Z","iopub.execute_input":"2022-12-10T20:34:12.575623Z","iopub.status.idle":"2022-12-10T20:34:12.889777Z","shell.execute_reply.started":"2022-12-10T20:34:12.575592Z","shell.execute_reply":"2022-12-10T20:34:12.888703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[train.BIRADS.isnull()]['cancer'].value_counts()[0]/train.cancer.value_counts()[0]","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:12.891158Z","iopub.execute_input":"2022-12-10T20:34:12.89157Z","iopub.status.idle":"2022-12-10T20:34:12.905842Z","shell.execute_reply.started":"2022-12-10T20:34:12.89154Z","shell.execute_reply":"2022-12-10T20:34:12.904525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 4.140% of patients breast was rated as normal and it belongs to cancer False patients\n\n* 28.230% of patients breast was rated as cancer negative.\n\n* 42.66% of cancer true patients, and 52.15% of cancer false patients BIRADS parameter is null.\n\n\n","metadata":{}},{"cell_type":"markdown","source":"# implant","metadata":{}},{"cell_type":"markdown","source":"implant - Whether or not the patient had breast implants. Site 1 only provides breast implant information at the patient level, not at the breast level.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,5))\n\nax = sns.countplot(data = train, x = 'implant', hue='site_id', palette='hot_r')\nt=train.shape[0];ax.set_title('implant Visualization', fontweight='bold')\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / t:.3f}%\\n';x = p.get_x() + p.get_width() / 2 ; y = p.get_height();\n    ax.annotate(percentage, (x, y), ha='center', va='center');ax.grid(axis='y', linestyle='-', alpha=0.4) \n    ax.legend(loc='upper right')\n    \nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:12.907181Z","iopub.execute_input":"2022-12-10T20:34:12.907628Z","iopub.status.idle":"2022-12-10T20:34:13.143552Z","shell.execute_reply.started":"2022-12-10T20:34:12.907595Z","shell.execute_reply":"2022-12-10T20:34:13.142505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(20,8))\n\nsns.countplot(data = train, x = 'implant', hue='site_id', palette='gnuplot',ax = ax[0])\nsns.countplot(data = train, x = 'implant', hue='cancer', palette='gnuplot',ax=ax[1])\nt=train.shape[0]\n\n    \nfor i, ax in enumerate(ax.flatten()):\n    ax.grid(axis='y', linestyle='-', alpha=0.4) \n    \n    if i==0:ax.set_title('implant vs sites1', fontweight='bold')\n    else:ax.set_title('implant vs cancer status',fontweight='bold')\n        \n        \n        \n    for p in ax.patches:\n        percentage = f'{100 * p.get_height() / t:.2f}%\\n'\n        ax.annotate(percentage, (p.get_x() + p.get_width() / 2,p.get_height()), ha='center', va='center')\n        ax.legend(loc='upper right')\n           \nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:13.14489Z","iopub.execute_input":"2022-12-10T20:34:13.145233Z","iopub.status.idle":"2022-12-10T20:34:13.53928Z","shell.execute_reply.started":"2022-12-10T20:34:13.145202Z","shell.execute_reply":"2022-12-10T20:34:13.538066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Implant is done only in site_id 1 and it is only 2.7%\n\n* Out of 2.7% implant done only 2.68% is having cancer status as False and 0.02% as True","metadata":{}},{"cell_type":"markdown","source":"# Density","metadata":{}},{"cell_type":"markdown","source":"density - A rating for how dense the breast tissue is, with **A being the least dense** and **D being the most dense**. Extremely dense tissue can make diagnosis more difficult. ","metadata":{}},{"cell_type":"code","source":"temp = train.copy();temp.density.fillna('Null', inplace=True)\n\nplt.figure(figsize=(15,5))\n\nax = sns.countplot(data = temp, x = 'density',order=['Null','A','B','C','D'],palette='Oranges')\nt=train.shape[0];ax.set_title('density Visualization', fontweight='bold')\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / t:.3f}%\\n';x = p.get_x() + p.get_width() / 2 ; y = p.get_height();\n    ax.annotate(percentage, (x, y), ha='center', va='center');ax.grid(axis='y', linestyle='-', alpha=0.4) \n    ax.legend(loc='upper right')\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-10T20:34:13.540979Z","iopub.execute_input":"2022-12-10T20:34:13.54131Z","iopub.status.idle":"2022-12-10T20:34:13.766658Z","shell.execute_reply.started":"2022-12-10T20:34:13.54128Z","shell.execute_reply":"2022-12-10T20:34:13.765306Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 46.1330% of paitents density parameter is null\n\n* 2.813% is having very dense breast tissue, which will be difficult to diaganosis","metadata":{}},{"cell_type":"markdown","source":"# machine_id","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,5))\n\nax = sns.countplot(data = temp, x = 'machine_id',hue='site_id', palette='gnuplot2')\nt=train.shape[0];ax.set_title('machine_id Visualization', fontweight='bold')\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / t:.3f}%\\n';x = p.get_x() + p.get_width() / 2 ; y = p.get_height();\n    ax.annotate(percentage, (x, y), ha='center', va='center');ax.grid(axis='y', linestyle='-', alpha=0.4) \n    ax.legend(loc='upper right')\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-10T20:34:13.768183Z","iopub.execute_input":"2022-12-10T20:34:13.768642Z","iopub.status.idle":"2022-12-10T20:34:14.121692Z","shell.execute_reply.started":"2022-12-10T20:34:13.768588Z","shell.execute_reply":"2022-12-10T20:34:14.120536Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Machine id 21, 29 and 48 are used in site_2 and rest machines in site_1\n\n* 79.708% of patients of site_1 is analyed using machine ID 49","metadata":{}},{"cell_type":"markdown","source":"# difficult_negative_case","metadata":{}},{"cell_type":"code","source":"train.difficult_negative_case.value_counts()\n\ntemp = train.copy();temp.density.fillna('Null', inplace=True)\n\nplt.figure(figsize=(15,5))\n\nax = sns.countplot(data = temp, x = 'difficult_negative_case',hue = 'density',palette='Oranges_r')\nt=train.shape[0];ax.set_title('density Visualization', fontweight='bold')\n\nfor p in ax.patches:\n    percentage = f'{100 * p.get_height() / t:.3f}%\\n';x = p.get_x() + p.get_width() / 2 ; y = p.get_height();\n    ax.annotate(percentage, (x, y), ha='center', va='center');ax.grid(axis='y', linestyle='-', alpha=0.4) \n    ax.legend(loc='upper right')\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-10T20:34:14.123331Z","iopub.execute_input":"2022-12-10T20:34:14.123783Z","iopub.status.idle":"2022-12-10T20:34:14.444852Z","shell.execute_reply.started":"2022-12-10T20:34:14.123741Z","shell.execute_reply":"2022-12-10T20:34:14.44252Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Image analysis","metadata":{}},{"cell_type":"markdown","source":"Refered Document : https://pydicom.github.io/pydicom/1.1/auto_examples/input_output/plot_read_dicom.html","metadata":{}},{"cell_type":"markdown","source":"## First Image in training data","metadata":{}},{"cell_type":"code","source":"train_images = gb.glob('/kaggle/input/rsna-breast-cancer-detection/train_images/10**/*.dcm', recursive = True)\nlen(train_images)\n\nfilename = train_images[0]\ndataset = dicom.dcmread(filename)\nprint(dataset)\nprint(__doc__)\n\nprint(\"Filename.........:\", filename);\nprint(\"Patient id.......:\", dataset.PatientID)\n\nif 'PixelData' in dataset:\n    rows = int(dataset.Rows)\n    cols = int(dataset.Columns)\n    print(\"Image size.......: {rows:d} x {cols:d}, {size:d} bytes\".format(\n        rows=rows, cols=cols, size=len(dataset.PixelData)))\n    if 'PixelSpacing' in dataset:\n        print(\"Pixel spacing....:\", dataset.PixelSpacing)\n\nprint(\"Slice location...:\", dataset.get('SliceLocation', \"(missing)\"))\n\nplt.imshow(dataset.pixel_array, cmap=plt.cm.bone)\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:14.450505Z","iopub.execute_input":"2022-12-10T20:34:14.451163Z","iopub.status.idle":"2022-12-10T20:34:19.799308Z","shell.execute_reply.started":"2022-12-10T20:34:14.451124Z","shell.execute_reply":"2022-12-10T20:34:19.798195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Image of Patient with Cancer False Patient with different color map","metadata":{}},{"cell_type":"code","source":"dcm_train_img_path = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\n\ndef train_images(patient_id , figsize,cmap):\n    fig, ax = plt.subplots(1, len(list(train[train.patient_id == patient_id].patient_id)), figsize=figsize);ax = ax.flatten()\n    \n    for i, dcm in enumerate(os.listdir(os.path.join(dcm_train_img_path, str(patient_id)))):\n        img = (dicom.dcmread(os.path.join(dcm_train_img_path, str(patient_id), dcm))).pixel_array\n        title = \"Patient_id : \" + str(patient_id) + ' ' + cmap + ', image - ' + str(i);ax[i].set_title(title)\n        if len(cmap)==0:ax[i].imshow(img)\n        else:ax[i].imshow(img, cmap=cmap);","metadata":{"execution":{"iopub.status.busy":"2022-12-10T20:34:19.800636Z","iopub.execute_input":"2022-12-10T20:34:19.800963Z","iopub.status.idle":"2022-12-10T20:34:19.809407Z","shell.execute_reply.started":"2022-12-10T20:34:19.800933Z","shell.execute_reply":"2022-12-10T20:34:19.808272Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_id = cancer_false_df['patient_id'].unique()[0]\ntrain_images(patient_id, (20,10),'')\ntrain_images(patient_id, (20,10),'bone')\ntrain_images(patient_id, (20,10),'gray')\ntrain_images(patient_id, (20,10),'inferno')\ntrain_images(patient_id, (20,10),'viridis')\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:34:19.810947Z","iopub.execute_input":"2022-12-10T20:34:19.811369Z","iopub.status.idle":"2022-12-10T20:35:30.600629Z","shell.execute_reply.started":"2022-12-10T20:34:19.811328Z","shell.execute_reply":"2022-12-10T20:35:30.599527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Image of Patient with Cancer True Patient from beginning and whose cancer status has not changed with different color map","metadata":{}},{"cell_type":"code","source":"patient_id = cancer_true_df['patient_id'].unique()[0]\n\ntrain_images(patient_id, (20,10),'')\ntrain_images(patient_id, (20,10),'bone')\ntrain_images(patient_id, (20,10),'gray')\ntrain_images(patient_id, (20,10),'inferno')\ntrain_images(patient_id, (20,10),'viridis')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-10T20:35:30.60196Z","iopub.execute_input":"2022-12-10T20:35:30.602317Z","iopub.status.idle":"2022-12-10T20:36:46.343428Z","shell.execute_reply.started":"2022-12-10T20:35:30.602284Z","shell.execute_reply":"2022-12-10T20:36:46.342219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"_kg_hide-input":true}}]}