{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":" !pip install matplotlib==3.5.2","metadata":{"executionInfo":{"elapsed":10634,"status":"ok","timestamp":1667947329496,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"vd7QZw5r3GLR","outputId":"5c97ecd8-73d7-4fbe-d0cf-4df4ce5ed419","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-01-08T12:05:51.522192Z","iopub.execute_input":"2023-01-08T12:05:51.522764Z","iopub.status.idle":"2023-01-08T12:06:04.595693Z","shell.execute_reply.started":"2023-01-08T12:05:51.522651Z","shell.execute_reply":"2023-01-08T12:06:04.594099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U pylibjpeg pylibjpeg-openjpeg pylibjpeg-libjpeg pydicom python-gdcm","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-01-08T12:06:04.59848Z","iopub.execute_input":"2023-01-08T12:06:04.598893Z","iopub.status.idle":"2023-01-08T12:06:16.501339Z","shell.execute_reply.started":"2023-01-08T12:06:04.598853Z","shell.execute_reply":"2023-01-08T12:06:16.500062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📌 Import libraries.\n\n","metadata":{"id":"zqrYqv734KAo"}},{"cell_type":"code","source":"# Core\nimport numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt \nimport seaborn as sns;\nfrom scipy import stats\nimport glob\nimport random\nimport datetime\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport plotly.io as pio\nimport gc\nimport os\nimport pickle\nfrom tqdm import tqdm\n# Core\nimport numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt \nimport seaborn as sns; sns.set(rc={'figure.figsize':[7,7]},font_scale=1.2)\nfrom datetime import date,timedelta\nfrom mpl_toolkits.mplot3d import Axes3D\nfrom mpl_toolkits.mplot3d import axes3d\nimport sklearn\nfrom sklearn.preprocessing import LabelEncoder,OneHotEncoder\nfrom tensorflow.keras.applications import  Xception,VGG16,InceptionResNetV2\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping ,ReduceLROnPlateau\nfrom tensorflow.keras.utils import to_categorical\nfrom keras.models import load_model\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Flatten, Dense,BatchNormalization,Dropout,Input\nfrom keras.models import Sequential, Model\nfrom keras.layers import Conv2D,GlobalMaxPooling2D\nfrom tensorflow.keras.applications import  Xception,VGG16,InceptionResNetV2\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n#cnn\nfrom tensorflow.keras import datasets, layers, models\nfrom keras.regularizers import l2 \n\nfrom keras.layers import Input, Conv2D, MaxPooling2D, UpSampling2D, concatenate, Conv2DTranspose, BatchNormalization, Dropout, Lambda\nfrom keras.engine.base_layer import Layer\n# Pre Processing\nfrom sklearn.impute import SimpleImputer \nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import MinMaxScaler\n# Regressors\nfrom sklearn.linear_model import Ridge\nfrom sklearn.linear_model import Lasso\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.ensemble import RandomForestRegressor,AdaBoostRegressor\nfrom sklearn.svm import SVR\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.preprocessing import PolynomialFeatures\nfrom sklearn.linear_model import LogisticRegression\n# Error Metrics \nfrom sklearn.metrics import r2_score #r2 square\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import plot_confusion_matrix ,classification_report\nfrom sklearn.metrics import accuracy_score, precision_score,recall_score,f1_score\nfrom IPython.display import clear_output, display_html\nimport os\nimport warnings\nfrom pathlib import Path\nimport gdcm\nimport pydicom as dicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\n#classefication\nfrom sklearn.svm import SVC\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn import tree\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.linear_model import SGDClassifier #stacstic gradient descent clasifeier\nimport graphviz\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier\n#crossvalidation\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import LeaveOneOut\nfrom sklearn.metrics import plot_confusion_matrix\n#clustring \nfrom sklearn.cluster import KMeans\nfrom sklearn.cluster import AgglomerativeClustering\n#hyper parameter tunning\nfrom sklearn.model_selection import GridSearchCV\n#pca\nfrom sklearn.decomposition import PCA\n#clustring\nfrom sklearn.cluster import KMeans\nfrom warnings import filterwarnings\nfilterwarnings(\"ignore\")","metadata":{"id":"1eb99db7","execution":{"iopub.status.busy":"2023-01-08T12:33:35.017449Z","iopub.execute_input":"2023-01-08T12:33:35.017979Z","iopub.status.idle":"2023-01-08T12:33:35.040143Z","shell.execute_reply.started":"2023-01-08T12:33:35.017939Z","shell.execute_reply":"2023-01-08T12:33:35.038966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed = 42\nnp.random.seed =seed","metadata":{"id":"4DTeuvL_TphS","execution":{"iopub.status.busy":"2023-01-08T12:06:20.014859Z","iopub.execute_input":"2023-01-08T12:06:20.015473Z","iopub.status.idle":"2023-01-08T12:06:20.021437Z","shell.execute_reply.started":"2023-01-08T12:06:20.015418Z","shell.execute_reply":"2023-01-08T12:06:20.019924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# 📌Business Goal \n \n**The goal of this competition is to identify cases of breast cancer in mammograms from screening exams. It is important to identify cases of cancer for obvious reasons, but false positives also have downsides for patients. As millions of women get mammograms each year, a useful machine learning tool could help a great many people.**\n\n**This competition uses a hidden test. When your submitted notebook is scored the actual test data (including a full length sample submission) will be made available to your notebook.**\n","metadata":{"id":"k8flQhplXsZS"}},{"cell_type":"markdown","source":"# 📌 About Dataset    \n\n[train/test]_images/[patient_id]/[image_id].dcm The mammograms, in dicom format. You can expect roughly 8,000 patients in the hidden test set. There are usually but not always 4 images per patient. Note that many of the images use the jpeg 2000 format which may you may need special libraries to load.\n\nsample_submission.csv A valid sample submission. Only the first few rows are available for download.\n\n[train/test].csv Metadata for each patient and image. Only the first few rows of the test set are available for download.\n\n1. **site_id** - ID code for the source hospital.\n2. **patient_id** - ID code for the patient.\n3. **image_id**- ID code for the image.\n4. **laterality** - Whether the image is of the left or right breast.\nview - The orientation of the image. The default for a screening exam is to capture two views per breast.\n5. **age** - The patient's age in years.\n6. **implant** - Whether or not the patient had breast implants. Site  only provides breast implant information at the patient level, not at the breast level.\n7. **density** - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult. Only provided for train.\n8. **machine_id** - An ID code for the imaging device.\n9. **cancer** - Whether or not the breast was positive for malignant \n10. **cancer**. The target value. Only provided for train.\n11. **biopsy** - Whether or not a follow-up biopsy was performed on the **breast**. Only provided for train.\n12. **invasive** - If the breast is positive for cancer, whether or not the cancer proved to be invasive. Only provided for train.\n13. **BIRADS** - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.\n14. **prediction_id** - The ID for the matching submission row. Multiple images will share the same prediction ID. Test only.\ndifficult_negative_case - True if the case was unusually difficult. Only provided for train.\n","metadata":{"id":"j8uR3Mzb_4tj"}},{"cell_type":"markdown","source":"#  📌Helper Function ⚒","metadata":{"id":"2zDVR6bc8DyI"}},{"cell_type":"code","source":"#convert data frame to slower case\ndef lowerCase(x):\n    return x.lower()\n\n#check duplicate data \ndef check_duplicate(df):\n    if df.duplicated().all():\n        return  'There are duplicate Data in Data Frame Nedded To be  removed . ' \n    else :\n        return 'Data Is clean ,No Duplicate Data Found .'\n# get label Name   \ndef get_Label(number):\n    labels = {0:'NORMAL', 1:'Cancer'}\n    return labels[number]\n\ndef calc_day_of_birth (day_num):\n    today = date.today() \n    birthDay = (today + timedelta(days=day_num)).strftime('%Y-%m-%d')\n    return birthDay\n    \ndef calc_day_of_employed(day_num):\n    today = date.today() \n    employedDay = (today + timedelta(days=day_num)).strftime('%Y-%m-%d')\n    result = 0\n    if employedDay > date.today().strftime('%Y-%m-%d') :\n         result = 0\n    else:\n         result = employedDay\n    return result\n\ndef calculate_age(born):\n    born = datetime.datetime.strptime(born, '%Y-%m-%d')\n    today = date.today()\n    return today.year - born.year - ((today.month, today.day) < (born.month, born.day))\n    \n    \ndef get_appartment(x):\n    if x == 'House / apartment' :\n       x= x.split(' /')[0]       \n    return x\n    \ndef get_ducational_type(x):\n    if x == 'Secondary / secondary special' :\n       x= x.split(' /')[0]       \n    return x\n\ndef get_label_for_data(x):\n    target = ''\n    if x in (2,3,4,5) :\n       target = 'YES' #risky\n    else:\n         target = 'NO'  #not risky\n\n    return target\n    #draw distplot for all numeric columns just pass numerical column\ndef all_distplot (numCol):\n    plt.figure(1 , figsize = (20 , 6))\n    n = 0 \n    for x in numCol:\n        n += 1\n        plt.subplot(1 , len(numCol) , n)\n        plt.subplots_adjust(hspace =0.5 , wspace = 0.5)\n        sns.distplot(df[x] , bins = 20)\n        plt.title('Distplot of {}'.format(x))\n    plt.show()    \n     \ndef box_plot(df):\n    i=1\n    plt.figure(figsize = (20,50))\n    for col in df.columns:\n        plt.subplot(round(len(df.columns)/3),3,i)\n        sns.boxplot(x = df[col], data = df,width = 0.5, fliersize = 3, linewidth = 1)\n        i+=1       \n\ndef numerical_plotting(df, col, title, symb):\n    fig, ax = plt.subplots(2, 1, sharex=True, figsize=(8,5),gridspec_kw={\"height_ratios\": (.2, .8)})\n    ax[0].set_title(title,fontsize=18)\n    sns.boxplot(x=col, data=df, ax=ax[0])\n    ax[0].set(yticks=[])\n    sns.distplot(df[col],kde=True)\n    plt.xticks(rotation=45)\n    ax[1].set_xlabel(col, fontsize=16)\n    plt.axvline(df[col].mean(), color='darkgreen', linewidth=2.2, label='mean=' + str(np.round(df[col].mean(),1)) + symb)\n    plt.axvline(df[col].median(), color='red', linewidth=2.2, label='median='+ str(np.round(df[col].median(),1)) + symb)\n    plt.axvline(df[col].mode()[0], color='purple', linewidth=2.2, label='mode='+ str(df[col].mode()[0]) + symb)\n    plt.legend(bbox_to_anchor=(1, 1.03), ncol=1, fontsize=17, fancybox=True, shadow=True, frameon=True)\n    plt.tight_layout()\n    plt.show()   \n\ndef categorical_plotting(df,col,title):\n    fig, ax = plt.subplots(figsize=(10,5))\n    ax=sns.countplot(x=col, data=df, palette='flare', order = df[col].value_counts().index)\n    ax.set_xticklabels(ax.get_xticklabels(), rotation=45)\n    ax.bar_label(ax.containers[0])\n    plt.title(title)\n    plt.show()\n\ndef plot_feature_importance (x,model,Model_name):\n    plt.figure(figsize=(15,20))\n    columns_list = x.columns\n    model.feature_names = columns_list\n    plt.barh(model.feature_names,sorted(model.coef_))\n    plt.xticks(rotation=45);\n    plt.title('Feature Importance'+ Model_name)\n    plt.xlabel('Feature Importance (%)')\n    plt.show()\ndef plot_feature_importance_2 (x,model,Model_name):\n    plt.figure(figsize=(15,20))\n    columns_list = x.columns\n    model.feature_names = columns_list\n    plt.barh(model.feature_names,sorted(model.feature_importances_))\n    plt.xticks(rotation=45);\n    plt.title('Feature Importance'+ Model_name)\n    plt.xlabel('Feature Importance (%)')\n    plt.show()\n\ndef lr_plot(df, col_x, col_y, leg):\n    slope, intercept, r_value, p_value, std_err = stats.linregress(df[col_x],df[col_y])\n    sns.regplot(x=col_x, y = col_y, data=df, color='#0d98ba', line_kws={'label':\"y={0:.1f}x+{1:.1f}\".format(slope,intercept)})\n    plt.legend(loc=leg, ncol=1, fontsize=15, fancybox=True, shadow=True, frameon=True)\n    plt.title(col_y + ' VS ' + col_x)\n    plt.show()\n\n    return slope, intercept\ndef average_plotting(df,col,output,number,title):\n    data_list = df[col].value_counts().index[:number].tolist()\n    plt.figure(figsize=(15,5))\n    ax=sns.barplot(x=col, y=output, data=df[df[col].isin(data_list)],order=data_list,palette='flare',ci=False,edgecolor=\"black\") \n    plt.xticks(rotation=45);\n    ax.bar_label(ax.containers[0])\n    plt.title(title)\n    plt.show()\ndef draw_unique_value (df,title):\n    plt.figure(figsize=(10,5))\n    plt.title(title)\n    unique_counts = df.nunique().to_dict()\n    ax = sns.barplot(list(unique_counts.keys()), list(unique_counts.values()),palette='flare')\n    ax.bar_label(ax.containers[0])\n    plt.plot()\n    \ndef freezing_layers(model_name):\n    for layer in model_name.layers:\n       layer.trainable = False     ","metadata":{"id":"nJ59u7qG_AFv","execution":{"iopub.status.busy":"2023-01-08T12:20:11.300929Z","iopub.execute_input":"2023-01-08T12:20:11.301485Z","iopub.status.idle":"2023-01-08T12:20:11.33964Z","shell.execute_reply.started":"2023-01-08T12:20:11.301441Z","shell.execute_reply":"2023-01-08T12:20:11.337837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  📌Reading Data","metadata":{"id":"pDFfcz3FbuGY"}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntest  = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\nsample_submission = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')","metadata":{"id":"7PRqZPU0aud1","execution":{"iopub.status.busy":"2023-01-08T12:06:20.218373Z","iopub.execute_input":"2023-01-08T12:06:20.21878Z","iopub.status.idle":"2023-01-08T12:06:20.370065Z","shell.execute_reply.started":"2023-01-08T12:06:20.218747Z","shell.execute_reply":"2023-01-08T12:06:20.368601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"executionInfo":{"elapsed":103,"status":"ok","timestamp":1667837663393,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"D-CzpOBQV_eS","outputId":"3e686952-60e7-4b81-8397-cc8763a7dcfa","execution":{"iopub.status.busy":"2023-01-08T12:06:20.371737Z","iopub.execute_input":"2023-01-08T12:06:20.372127Z","iopub.status.idle":"2023-01-08T12:06:20.411586Z","shell.execute_reply.started":"2023-01-08T12:06:20.372093Z","shell.execute_reply":"2023-01-08T12:06:20.410346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"executionInfo":{"elapsed":89,"status":"ok","timestamp":1667837663395,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"1Ow4JZmSV_iQ","outputId":"9488b8a9-91dc-4060-f019-1ed6396c972d","execution":{"iopub.status.busy":"2023-01-08T12:06:20.413226Z","iopub.execute_input":"2023-01-08T12:06:20.414514Z","iopub.status.idle":"2023-01-08T12:06:20.429194Z","shell.execute_reply.started":"2023-01-08T12:06:20.414464Z","shell.execute_reply":"2023-01-08T12:06:20.427884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission","metadata":{"executionInfo":{"elapsed":75,"status":"ok","timestamp":1667837663401,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"xj6A9aUwYQyk","outputId":"4b1633ef-422c-4754-fdfd-d2ba71d9d76e","execution":{"iopub.status.busy":"2023-01-08T12:06:20.430864Z","iopub.execute_input":"2023-01-08T12:06:20.431817Z","iopub.status.idle":"2023-01-08T12:06:20.446365Z","shell.execute_reply.started":"2023-01-08T12:06:20.43177Z","shell.execute_reply":"2023-01-08T12:06:20.445034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h4  id=\"1.1-|-set dataset columns to lower\"><b>1.1 <span style=\"color:#4a8fdd\">|</span> set dataset columns to lower</b></h4>\n","metadata":{"id":"iuhuJe6BY0c9"}},{"cell_type":"code","source":"#set column name to lowerCase\ndf = df.rename(columns=str.lower)\ntest  = test.rename(columns=str.lower)","metadata":{"id":"vCr1z70ZY1z0","execution":{"iopub.status.busy":"2023-01-08T12:06:20.451724Z","iopub.execute_input":"2023-01-08T12:06:20.452109Z","iopub.status.idle":"2023-01-08T12:06:20.462138Z","shell.execute_reply.started":"2023-01-08T12:06:20.452076Z","shell.execute_reply":"2023-01-08T12:06:20.461052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h4 id=\"1.2-|-dataset shape\"><b>1.2 <span style=\"color:#4a8fdd\">|</span> dataset shape</b></h4>\n\n\n\n","metadata":{"id":"2mSMNR4qc8_H"}},{"cell_type":"code","source":"print(\"The number of rows in  Train data is {} , \\nThe number of columns in  data is {}\".format(df.shape[0], df.shape[1])) \nprint(\"The number of rows in  Test data is {} , \\nThe number of columns in  data is {}\".format(test.shape[0], test.shape[1])) ","metadata":{"executionInfo":{"elapsed":74,"status":"ok","timestamp":1667837663404,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"6b6pvW8NY2BD","outputId":"38686600-0355-4c9d-cef3-9bfd8591c316","execution":{"iopub.status.busy":"2023-01-08T12:06:20.464134Z","iopub.execute_input":"2023-01-08T12:06:20.464629Z","iopub.status.idle":"2023-01-08T12:06:20.474012Z","shell.execute_reply.started":"2023-01-08T12:06:20.464586Z","shell.execute_reply":"2023-01-08T12:06:20.472988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📌Data Cleaning\n\n*   check column type.\n*   drop un necessary column.\n*   check duplicate data \n*   check  missing value.\n*   dealing with missing value\n\n","metadata":{"id":"y4zlF2xKKKJt"}},{"cell_type":"markdown","source":"<h4 id=\"2.1-|-dataset information\"><b>2.1<span style=\"color:#4a8fdd\">|</span> dataset information</b></h4>","metadata":{"id":"YlLZiCvnhC0b"}},{"cell_type":"code","source":"df.info()","metadata":{"executionInfo":{"elapsed":19,"status":"ok","timestamp":1667410112246,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"BLg8AOG8gBcX","outputId":"60ae0033-568d-4a4c-e616-6ae633be742f","execution":{"iopub.status.busy":"2023-01-08T12:06:20.475571Z","iopub.execute_input":"2023-01-08T12:06:20.477617Z","iopub.status.idle":"2023-01-08T12:06:20.518285Z","shell.execute_reply.started":"2023-01-08T12:06:20.477569Z","shell.execute_reply":"2023-01-08T12:06:20.517055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numCol  = [col for col in df.columns if  df[col].dtype != \"O\"]\nnumCol","metadata":{"executionInfo":{"elapsed":374,"status":"ok","timestamp":1667761689695,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"VHnRlAUbNnIb","outputId":"9f1c6998-e551-4c92-e179-0e5915b484b6","execution":{"iopub.status.busy":"2023-01-08T12:06:20.519989Z","iopub.execute_input":"2023-01-08T12:06:20.520376Z","iopub.status.idle":"2023-01-08T12:06:20.529029Z","shell.execute_reply.started":"2023-01-08T12:06:20.520341Z","shell.execute_reply":"2023-01-08T12:06:20.527763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catColumn  = [col for col in df.columns if  df[col].dtype == \"O\"]\ncatColumn","metadata":{"executionInfo":{"elapsed":12,"status":"ok","timestamp":1667761689699,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"gyOpiM6ljhTE","outputId":"46efabad-4b9e-448b-d554-5a6640076f2d","execution":{"iopub.status.busy":"2023-01-08T12:06:20.530892Z","iopub.execute_input":"2023-01-08T12:06:20.531392Z","iopub.status.idle":"2023-01-08T12:06:20.542794Z","shell.execute_reply.started":"2023-01-08T12:06:20.531348Z","shell.execute_reply":"2023-01-08T12:06:20.541547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#set option to data frame to show clear values\npd.set_option('display.float_format', lambda x: '%.5f' % x)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:20.544314Z","iopub.execute_input":"2023-01-08T12:06:20.544722Z","iopub.status.idle":"2023-01-08T12:06:20.554148Z","shell.execute_reply.started":"2023-01-08T12:06:20.544685Z","shell.execute_reply":"2023-01-08T12:06:20.552501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe().T","metadata":{"executionInfo":{"elapsed":1733,"status":"ok","timestamp":1667415444004,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"50MCwrimekaL","outputId":"3078888d-1be1-44d2-df9a-2507989f5c06","execution":{"iopub.status.busy":"2023-01-08T12:06:20.556616Z","iopub.execute_input":"2023-01-08T12:06:20.557038Z","iopub.status.idle":"2023-01-08T12:06:20.621153Z","shell.execute_reply.started":"2023-01-08T12:06:20.557001Z","shell.execute_reply":"2023-01-08T12:06:20.619966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h4 id=\"2.3-|-dataset statistics\"><b>2.3 <span style=\"color:#4a8fdd\">|</span> dataset statistics</b></h4>\n\n\n","metadata":{"id":"2O3iIDEcqaTb"}},{"cell_type":"code","source":"#check duplicate data \ncheck_duplicate(df)","metadata":{"executionInfo":{"elapsed":5385,"status":"ok","timestamp":1667325986658,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"BKItthh0rflm","outputId":"b47b60f5-1f64-4eae-91d3-d51a74ed8693","execution":{"iopub.status.busy":"2023-01-08T12:06:20.622939Z","iopub.execute_input":"2023-01-08T12:06:20.623417Z","iopub.status.idle":"2023-01-08T12:06:20.664478Z","shell.execute_reply.started":"2023-01-08T12:06:20.623381Z","shell.execute_reply":"2023-01-08T12:06:20.663097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check missing Value\ndf.isnull().sum().sort_values(ascending=False)","metadata":{"executionInfo":{"elapsed":1782,"status":"ok","timestamp":1667325988427,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"qh08DYx3sKK3","outputId":"175e2222-9c3c-46fe-cfc6-fc969a955760","execution":{"iopub.status.busy":"2023-01-08T12:06:20.665865Z","iopub.execute_input":"2023-01-08T12:06:20.666457Z","iopub.status.idle":"2023-01-08T12:06:20.689566Z","shell.execute_reply.started":"2023-01-08T12:06:20.666413Z","shell.execute_reply":"2023-01-08T12:06:20.688378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols =df.columns\nsns.heatmap(df[cols].isnull(), cmap='viridis')","metadata":{"executionInfo":{"elapsed":56841,"status":"ok","timestamp":1667318998345,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"NdrziDYEvNgX","outputId":"c7475ef5-03cd-4746-cc75-c3737d64ba3f","execution":{"iopub.status.busy":"2023-01-08T12:06:20.691078Z","iopub.execute_input":"2023-01-08T12:06:20.691523Z","iopub.status.idle":"2023-01-08T12:06:21.941108Z","shell.execute_reply.started":"2023-01-08T12:06:20.691489Z","shell.execute_reply":"2023-01-08T12:06:21.939882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h4 id=\"2.4-|-handel missing data\"><b>2.4 <span style=\"color:#4a8fdd\">|</span> handel missing data</b></h4>\n\n\n","metadata":{"id":"aQmsxlgztfv9"}},{"cell_type":"code","source":"#fill null data with median for numerical column  \ndf['birads'].fillna(df['birads'].median(), inplace=True)\ndf['age'].fillna(df['age'].median(), inplace=True)\n#fill null data with mode  for categourical column\ndf['density'].fillna(df['density'].mode()[0], inplace=True)\n","metadata":{"id":"biil-tXJmmwF","execution":{"iopub.status.busy":"2023-01-08T12:06:21.943167Z","iopub.execute_input":"2023-01-08T12:06:21.943691Z","iopub.status.idle":"2023-01-08T12:06:21.961473Z","shell.execute_reply.started":"2023-01-08T12:06:21.943645Z","shell.execute_reply":"2023-01-08T12:06:21.960137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols =df.columns\nsns.heatmap(df[cols].isnull(), cmap='viridis')","metadata":{"executionInfo":{"elapsed":51863,"status":"ok","timestamp":1667837721687,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"JyyUCKJBnbam","outputId":"5f4636be-1088-40d1-b22c-817b9939d2b3","execution":{"iopub.status.busy":"2023-01-08T12:06:21.962874Z","iopub.execute_input":"2023-01-08T12:06:21.963881Z","iopub.status.idle":"2023-01-08T12:06:23.263683Z","shell.execute_reply.started":"2023-01-08T12:06:21.963834Z","shell.execute_reply":"2023-01-08T12:06:23.262184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"executionInfo":{"elapsed":12,"status":"ok","timestamp":1667319000179,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"59CZacZotwTN","outputId":"84943f49-adee-4080-ebf7-efe0a4c2bbe1","execution":{"iopub.status.busy":"2023-01-08T12:06:23.265515Z","iopub.execute_input":"2023-01-08T12:06:23.266057Z","iopub.status.idle":"2023-01-08T12:06:23.273768Z","shell.execute_reply.started":"2023-01-08T12:06:23.266011Z","shell.execute_reply":"2023-01-08T12:06:23.272412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"text_cell_render border-box-sizing rendered_html\">\n<div style=\"border-radius: 10px;\n            border : black solid;\n            background-color: #34baeb;\n            font-size:110%;\n            text-align: left\">\n\n<h3 style=\"; border:0; border-radius: 10px; font-weight: bold; color:black\"><center> Dataset Basic Informations</center></h3>\n<p>● The dataset consists of 54706 rows and 14 columns. </p>\n<p>● We have 11 column as Numericaland floot column and 3 categorical column   .</p>\n<p>● There is 3 null data inside columns and solved .</p>\n<p>● No duplicate data in dataset .</p>\n\n</div>\n</div>","metadata":{"id":"b_kvPC_rtsYU"}},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:23.275647Z","iopub.execute_input":"2023-01-08T12:06:23.276109Z","iopub.status.idle":"2023-01-08T12:06:23.295017Z","shell.execute_reply.started":"2023-01-08T12:06:23.276066Z","shell.execute_reply":"2023-01-08T12:06:23.293595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:23.296419Z","iopub.execute_input":"2023-01-08T12:06:23.296812Z","iopub.status.idle":"2023-01-08T12:06:23.305746Z","shell.execute_reply.started":"2023-01-08T12:06:23.296776Z","shell.execute_reply":"2023-01-08T12:06:23.304593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_duplicate(test)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:23.307245Z","iopub.execute_input":"2023-01-08T12:06:23.307618Z","iopub.status.idle":"2023-01-08T12:06:23.321731Z","shell.execute_reply.started":"2023-01-08T12:06:23.307587Z","shell.execute_reply":"2023-01-08T12:06:23.320359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:23.323653Z","iopub.execute_input":"2023-01-08T12:06:23.324126Z","iopub.status.idle":"2023-01-08T12:06:23.33827Z","shell.execute_reply.started":"2023-01-08T12:06:23.324082Z","shell.execute_reply":"2023-01-08T12:06:23.336882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"text_cell_render border-box-sizing rendered_html\">\n<div style=\"border-radius: 10px;\n            border : black solid;\n            background-color: #34baeb;\n            font-size:110%;\n            text-align: left\">\n\n<h3 style=\"; border:0; border-radius: 10px; font-weight: bold; color:black\"><center> Test Dataset Basic Informations</center></h3>\n<p>● The dataset consists of 4 rows and 9 columns. </p>\n<p>● We have 6 column as Numericaland floot column and 3 categorical column   .</p>\n<p>● There is NO null data inside columns  .</p>\n<p>● No duplicate data in Test dataset .</p>\n<p>● shape od data is not equal with train data set ! .</p>\n\n</div>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# 📌Data Analaysis & Visualization\nin this part we will analays and versialize each part of data to be in near step from our goal then pased on deployed models we will sense best factior that affect on our bussiness goal\n","metadata":{"id":"zYgUE1GFu9w9"}},{"cell_type":"markdown","source":"\n<h4 id=\"3.3-|- distplot for neumerical data\"><b>3.1 <span style=\"color:#4a8fdd\">|</span> distplot for neumerical data</b></h4>\n\n","metadata":{"id":"1L22snQjvGnn"}},{"cell_type":"code","source":"numCol","metadata":{"executionInfo":{"elapsed":351,"status":"ok","timestamp":1667325994539,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"h2uySeIhvNeI","outputId":"fca4a8cd-6848-4cf9-dc9f-d57850e21c62","execution":{"iopub.status.busy":"2023-01-08T12:06:23.340063Z","iopub.execute_input":"2023-01-08T12:06:23.341067Z","iopub.status.idle":"2023-01-08T12:06:23.352245Z","shell.execute_reply.started":"2023-01-08T12:06:23.34102Z","shell.execute_reply":"2023-01-08T12:06:23.350996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_plotting(df,'age','age Distributionting','  ')","metadata":{"executionInfo":{"elapsed":17677,"status":"ok","timestamp":1667248282687,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"lSjcszAlvXd6","outputId":"4f15e723-f3e5-4459-f666-d6dc1a187ac1","execution":{"iopub.status.busy":"2023-01-08T12:06:23.357512Z","iopub.execute_input":"2023-01-08T12:06:23.358006Z","iopub.status.idle":"2023-01-08T12:06:24.2614Z","shell.execute_reply.started":"2023-01-08T12:06:23.357943Z","shell.execute_reply":"2023-01-08T12:06:24.260449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['age'].value_counts().sort_values(ascending=False).head(5)","metadata":{"executionInfo":{"elapsed":264,"status":"ok","timestamp":1667249044997,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"X1UijtTgyV_4","outputId":"4f4a9d4b-c2c0-4353-b49f-2eca52736e94","execution":{"iopub.status.busy":"2023-01-08T12:06:24.262991Z","iopub.execute_input":"2023-01-08T12:06:24.263741Z","iopub.status.idle":"2023-01-08T12:06:24.274962Z","shell.execute_reply.started":"2023-01-08T12:06:24.263705Z","shell.execute_reply":"2023-01-08T12:06:24.273641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<h4 id=\"3.4-|- Histogram\"><b>3.2<span style=\"color:#4a8fdd\">|</span>    Histogram </b></h4>\n","metadata":{"id":"AAmmg4Bm0BUI"}},{"cell_type":"code","source":"#diplay histogram for age column\nsns.histplot( x = df[\"age\"], bins = 20, kde = True, color = \"#D63913\").set(title = \"Distribution of age variable\");","metadata":{"executionInfo":{"elapsed":39229,"status":"ok","timestamp":1667251127847,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"YCclvwDG0Aer","outputId":"c3742e7f-d6de-41e5-e8f0-166dfcc5ae5b","execution":{"iopub.status.busy":"2023-01-08T12:06:24.276677Z","iopub.execute_input":"2023-01-08T12:06:24.277077Z","iopub.status.idle":"2023-01-08T12:06:24.849407Z","shell.execute_reply.started":"2023-01-08T12:06:24.27704Z","shell.execute_reply":"2023-01-08T12:06:24.848084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get age per patient \nagePerPatient=df.groupby('patient_id').agg({'age': lambda age:age.unique()}) \nfig, ax = plt.subplots(1,2,figsize=(20,5))\nsns.histplot(agePerPatient, color='#D63913', bins=60, ax=ax[0])\nax[0].set_title('Age Distribution Per Patient');\nsns.violinplot(df.cancer, df.age, ax=ax[1],palette='flare');","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:24.850837Z","iopub.execute_input":"2023-01-08T12:06:24.851211Z","iopub.status.idle":"2023-01-08T12:06:26.010005Z","shell.execute_reply.started":"2023-01-08T12:06:24.85118Z","shell.execute_reply":"2023-01-08T12:06:26.008811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()[['age']].T","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:26.01199Z","iopub.execute_input":"2023-01-08T12:06:26.012484Z","iopub.status.idle":"2023-01-08T12:06:26.063006Z","shell.execute_reply.started":"2023-01-08T12:06:26.012428Z","shell.execute_reply":"2023-01-08T12:06:26.061687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get how many image per patient \ndf.groupby('patient_id').agg({'image_id': lambda image_id:image_id.count()})   ","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:26.064448Z","iopub.execute_input":"2023-01-08T12:06:26.064799Z","iopub.status.idle":"2023-01-08T12:06:26.276873Z","shell.execute_reply.started":"2023-01-08T12:06:26.064768Z","shell.execute_reply":"2023-01-08T12:06:26.275536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get total number of patient \ndf.patient_id.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:01:59.145408Z","iopub.execute_input":"2023-01-08T12:01:59.146333Z","iopub.status.idle":"2023-01-08T12:01:59.154348Z","shell.execute_reply.started":"2023-01-08T12:01:59.146289Z","shell.execute_reply":"2023-01-08T12:01:59.152935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"text_cell_render border-box-sizing rendered_html\">\n<div style=\"border-radius: 10px;\n            border : black solid;\n            background-color: #34baeb;\n            font-size:110%;\n            text-align: left\">\n\n<h3 style=\"; border:0; border-radius: 10px; font-weight: bold; color:black\"><center> Basic Numeric analaysis</center></h3>\n<p>● Data is  Normaly Distributed</p>\n<p>● most of patient has age bigger than 40 year </p>\n <p>● mean of patient ages is 59 year</p>\n <p>● most of patient has 4 image </p>\n\n</div>\n</div>","metadata":{"id":"ns6zoiFO-Rl9"}},{"cell_type":"markdown","source":"<h4 id=\"3.5-|- pi Plot\"><b>3.3 <span style=\"color:#4a8fdd\">|</span>  pi Plot </b></h4>","metadata":{}},{"cell_type":"code","source":"categorical_plotting(df,'cancer','total count of cancer per patient')\ndf[\"cancer\"].value_counts().plot.pie( autopct='%1.3f%%', shadow = True);","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:26.278564Z","iopub.execute_input":"2023-01-08T12:06:26.279051Z","iopub.status.idle":"2023-01-08T12:06:26.634557Z","shell.execute_reply.started":"2023-01-08T12:06:26.279001Z","shell.execute_reply":"2023-01-08T12:06:26.632963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_plotting(df,'biopsy','total count of biopsy per patient')\ndf[\"biopsy\"].value_counts().plot.pie( autopct='%1.3f%%', shadow = True);","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:26.636836Z","iopub.execute_input":"2023-01-08T12:06:26.637686Z","iopub.status.idle":"2023-01-08T12:06:27.047187Z","shell.execute_reply.started":"2023-01-08T12:06:26.637621Z","shell.execute_reply":"2023-01-08T12:06:27.04522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_plotting(df,'invasive','total count of invasive per patient')\ndf[\"invasive\"].value_counts().plot.pie( autopct='%1.3f%%', shadow = True);","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:27.049745Z","iopub.execute_input":"2023-01-08T12:06:27.05103Z","iopub.status.idle":"2023-01-08T12:06:27.446832Z","shell.execute_reply.started":"2023-01-08T12:06:27.05096Z","shell.execute_reply":"2023-01-08T12:06:27.445102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_plotting(df,'birads','total count of birads per patient')\ndf[\"birads\"].value_counts().plot.pie( autopct='%1.3f%%', shadow = True);","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:02.067293Z","iopub.execute_input":"2023-01-08T12:02:02.068702Z","iopub.status.idle":"2023-01-08T12:02:02.489931Z","shell.execute_reply.started":"2023-01-08T12:02:02.068618Z","shell.execute_reply":"2023-01-08T12:02:02.487765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_plotting(df,'difficult_negative_case','total count of difficult_negative_case per patient')\ndf[\"difficult_negative_case\"].value_counts().plot.pie( autopct='%1.3f%%', shadow = True);","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:02.493434Z","iopub.execute_input":"2023-01-08T12:02:02.494153Z","iopub.status.idle":"2023-01-08T12:02:02.905219Z","shell.execute_reply.started":"2023-01-08T12:02:02.494088Z","shell.execute_reply":"2023-01-08T12:02:02.903552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_plotting(df,'laterality','total count of laterality per patient')\ndf[\"laterality\"].value_counts().plot.pie( autopct='%1.3f%%', shadow = True);","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:02.907829Z","iopub.execute_input":"2023-01-08T12:02:02.908919Z","iopub.status.idle":"2023-01-08T12:02:03.305198Z","shell.execute_reply.started":"2023-01-08T12:02:02.908849Z","shell.execute_reply":"2023-01-08T12:02:03.303495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_plotting(df,'view','total count of view per patient')\ndf[\"view\"].value_counts().plot.pie( autopct='%1.3f%%', shadow = True);","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:03.307876Z","iopub.execute_input":"2023-01-08T12:02:03.308963Z","iopub.status.idle":"2023-01-08T12:02:03.820998Z","shell.execute_reply.started":"2023-01-08T12:02:03.308894Z","shell.execute_reply":"2023-01-08T12:02:03.819357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_plotting(df,'density','total count of density per patient')\ndf[\"density\"].value_counts().plot.pie( autopct='%1.3f%%', shadow = True);","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:03.826003Z","iopub.execute_input":"2023-01-08T12:02:03.827106Z","iopub.status.idle":"2023-01-08T12:02:04.479803Z","shell.execute_reply.started":"2023-01-08T12:02:03.827036Z","shell.execute_reply":"2023-01-08T12:02:04.478237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"text_cell_render border-box-sizing rendered_html\">\n<div style=\"border-radius: 10px;\n            border : black solid;\n            background-color: #34baeb;\n            font-size:110%;\n            text-align: left\">\n\n<h3 style=\"; border:0; border-radius: 10px; font-weight: bold; color:black\"><center> Basic Categorical  analaysis</center></h3>\n<p>●97.88%  of patients  has canser whivh mean our dayt set is not Balanced</p>\n<p>● 94.57%  of patients  has not a follow-up biopsy was performed on the breast)</p>\n <p>●98.5 % of patients  has negative invasive </p>\n <p>● 80% of patients  breast required follow-up  ,15%  of patients was rated as negative for cancer ,4% breast  is normal</p>\n <p>● laterality  has equal percentage  for two types of patient (50%)</p>\n<p>● there are 4 types of denisty  (A,B,C,D), 62% of patient has density B ,22% has C denisty </p>\n\n</div>\n</div>","metadata":{}},{"cell_type":"markdown","source":"<h4 id=\"3.5-|- Histogram\"><b>3.4 <span style=\"color:#4a8fdd\">|</span>    Box Blot </b></h4>","metadata":{"id":"5uMZ5CQa7U84"}},{"cell_type":"code","source":" \nfig, axes = plt.subplots(1, 2, figsize = (25, 7))\n\nsns.boxplot(ax = axes[0], x = \"cancer\", y = \"age\", data = df, width = 0.7, orient = \"v\", fliersize = 5,\n            saturation = 1, linewidth = 3,palette='flare');\naxes[1].set_title(\"relationship between cancer and  age variables\");\n\nsns.boxplot(ax = axes[1], x = \"biopsy\", y = \"age\", data = df, width = 0.7, orient = \"v\", fliersize = 5,\n            saturation = 1, linewidth = 3,palette='flare')\naxes[0].set_title(\"relationship between biopsy and age variables\"); ","metadata":{"executionInfo":{"elapsed":3231,"status":"ok","timestamp":1667319052276,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"dzuNQe0t0G1Q","outputId":"91a9ef14-0f23-44fb-ee70-77c5773069ca","execution":{"iopub.status.busy":"2023-01-08T12:02:04.482328Z","iopub.execute_input":"2023-01-08T12:02:04.48334Z","iopub.status.idle":"2023-01-08T12:02:04.886669Z","shell.execute_reply.started":"2023-01-08T12:02:04.483276Z","shell.execute_reply":"2023-01-08T12:02:04.885363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" \nfig, axes = plt.subplots(1, 2, figsize = (25, 7))\n\nsns.boxplot(ax = axes[0], x = \"invasive\", y = \"age\", data = df, width = 0.7, orient = \"v\", fliersize = 5,\n            saturation = 1, linewidth = 3,palette='flare');\naxes[1].set_title(\"relationship between invasive and  age variables\");\n\nsns.boxplot(ax = axes[1], x = \"invasive\", y = \"age\", data = df, width = 0.7, orient = \"v\", fliersize = 5,\n            saturation = 1, linewidth = 3,palette='flare')\naxes[0].set_title(\"relationship between invasive and age variables\"); ","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:04.888363Z","iopub.execute_input":"2023-01-08T12:02:04.889655Z","iopub.status.idle":"2023-01-08T12:02:05.27892Z","shell.execute_reply.started":"2023-01-08T12:02:04.889612Z","shell.execute_reply":"2023-01-08T12:02:05.277616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" \nfig, axes = plt.subplots(1, 2, figsize = (25, 7))\n\nsns.boxplot(ax = axes[0], x = \"birads\", y = \"age\", data = df, width = 0.7, orient = \"v\", fliersize = 5,\n            saturation = 1, linewidth = 3,palette='flare');\naxes[1].set_title(\"relationship between birads and  age variables\");\n\nsns.boxplot(ax = axes[1], x = \"implant\", y = \"age\", data = df, width = 0.7, orient = \"v\", fliersize = 5,\n            saturation = 1, linewidth = 3,palette='flare')\naxes[0].set_title(\"relationship between implant and age variables\"); ","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:05.28105Z","iopub.execute_input":"2023-01-08T12:02:05.28156Z","iopub.status.idle":"2023-01-08T12:02:05.685603Z","shell.execute_reply.started":"2023-01-08T12:02:05.281513Z","shell.execute_reply":"2023-01-08T12:02:05.684703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" \nfig, axes = plt.subplots(1, 2, figsize = (25, 7))\n\nsns.boxplot(ax = axes[0], x = \"difficult_negative_case\", y = \"age\", data = df, width = 0.7, orient = \"v\", fliersize = 5,\n            saturation = 1, linewidth = 3,palette='flare');\naxes[1].set_title(\"relationship between difficult_negative_case and  age variables\");\n\nsns.boxplot(ax = axes[1], x = \"laterality\", y = \"age\", data = df, width = 0.7, orient = \"v\", fliersize = 5,\n            saturation = 1, linewidth = 3,palette='flare')\naxes[0].set_title(\"relationship between laterality and age variables\"); ","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:05.687438Z","iopub.execute_input":"2023-01-08T12:02:05.688379Z","iopub.status.idle":"2023-01-08T12:02:06.097384Z","shell.execute_reply.started":"2023-01-08T12:02:05.688329Z","shell.execute_reply":"2023-01-08T12:02:06.096206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" \nfig, axes = plt.subplots(1, 2, figsize = (25, 7))\n\nsns.boxplot(ax = axes[0], x = \"view\", y = \"age\", data = df, width = 0.7, orient = \"v\", fliersize = 5,\n            saturation = 1, linewidth = 3,palette='flare');\naxes[1].set_title(\"relationship between view and  age variables\");\n\nsns.boxplot(ax = axes[1], x = \"density\", y = \"age\", data = df, width = 0.7, orient = \"v\", fliersize = 5,\n            saturation = 1, linewidth = 3,palette='flare')\naxes[0].set_title(\"relationship between density and age variables\"); ","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:06.100508Z","iopub.execute_input":"2023-01-08T12:02:06.100872Z","iopub.status.idle":"2023-01-08T12:02:06.631915Z","shell.execute_reply.started":"2023-01-08T12:02:06.100837Z","shell.execute_reply":"2023-01-08T12:02:06.63045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()","metadata":{"executionInfo":{"elapsed":1486,"status":"ok","timestamp":1667761327252,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"ljqw2WU_6x9G","outputId":"3cdb8cb0-44d1-4a7a-a29f-cc36a4a285e3","execution":{"iopub.status.busy":"2023-01-08T12:02:06.633416Z","iopub.execute_input":"2023-01-08T12:02:06.63378Z","iopub.status.idle":"2023-01-08T12:02:06.673651Z","shell.execute_reply.started":"2023-01-08T12:02:06.633738Z","shell.execute_reply":"2023-01-08T12:02:06.672308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = [20, 10], facecolor = 'white')\nsns.heatmap(df.corr(),annot=True)","metadata":{"executionInfo":{"elapsed":2480,"status":"ok","timestamp":1667414237994,"user":{"displayName":"gaber geka","userId":"15703872159436343604"},"user_tz":-120},"id":"6HKN3z06oTR8","outputId":"8293be1b-612d-46f1-aae9-9372a86ca794","execution":{"iopub.status.busy":"2023-01-08T12:02:06.675245Z","iopub.execute_input":"2023-01-08T12:02:06.676385Z","iopub.status.idle":"2023-01-08T12:02:07.627171Z","shell.execute_reply.started":"2023-01-08T12:02:06.676342Z","shell.execute_reply":"2023-01-08T12:02:07.62586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<div class=\"text_cell_render border-box-sizing rendered_html\">\n<div style=\"border-radius: 10px;\n            border : black solid;\n            background-color: #34baeb;\n            font-size:110%;\n            text-align: left\">\n\n<h3 style=\"; border:0; border-radius: 10px; font-weight: bold; color:black\"><center>Big Attention</center></h3>\n<p>●  Multicollinearity   Not  detected    </p>\n    <p>● ther are negative  corrolation detected but in small values </p>\n\n</div>\n</div>","metadata":{"id":"JaPrC741fli5"}},{"cell_type":"markdown","source":"# 📌Dealing with images","metadata":{}},{"cell_type":"code","source":"train_path = '/kaggle/input/rsna-breast-cancer-detection/train_images/'\ntest_path = '/kaggle/input/rsna-breast-cancer-detection/test_images/'","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:38.760041Z","iopub.execute_input":"2023-01-08T12:06:38.760491Z","iopub.status.idle":"2023-01-08T12:06:38.765428Z","shell.execute_reply.started":"2023-01-08T12:06:38.760457Z","shell.execute_reply":"2023-01-08T12:06:38.764551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#add image path for any dcfile to data fram\ndf['image_path']   = str(train_path)+ df.patient_id.astype(str)+'/'+df.image_id.astype(str)+'.dcm' \ntest['image_path'] = str(test_path)+ test.patient_id.astype(str)+'/'+test.image_id.astype(str)+'.dcm'  ","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:38.965239Z","iopub.execute_input":"2023-01-08T12:06:38.965988Z","iopub.status.idle":"2023-01-08T12:06:39.082939Z","shell.execute_reply.started":"2023-01-08T12:06:38.96594Z","shell.execute_reply":"2023-01-08T12:06:39.08169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def image_name(x):\n#         imagename =x.split('/')[-1][:-4]\n#         return imagename\n    ","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:39.196608Z","iopub.execute_input":"2023-01-08T12:06:39.197064Z","iopub.status.idle":"2023-01-08T12:06:39.202778Z","shell.execute_reply.started":"2023-01-08T12:06:39.19703Z","shell.execute_reply":"2023-01-08T12:06:39.201455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df['new_path']= df['image_path'].apply(image_name)\n# test['new_path']=test['image_path'].apply(image_name)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T10:43:06.798494Z","iopub.execute_input":"2023-01-08T10:43:06.798957Z","iopub.status.idle":"2023-01-08T10:43:06.803346Z","shell.execute_reply.started":"2023-01-08T10:43:06.798919Z","shell.execute_reply":"2023-01-08T10:43:06.802379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['new_path']     = '/kaggle/working/output/train/'+df.patient_id.astype(str)+'_'+ df.image_id.astype(str)+'.png'\ntest['new_path']     = '/kaggle/working/output/test/'+test.patient_id.astype(str)+'_'+ test.image_id.astype(str)+'.png'","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:42.408123Z","iopub.execute_input":"2023-01-08T12:06:42.408609Z","iopub.status.idle":"2023-01-08T12:06:42.522093Z","shell.execute_reply.started":"2023-01-08T12:06:42.40857Z","shell.execute_reply":"2023-01-08T12:06:42.520759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📌Patient Images","metadata":{}},{"cell_type":"code","source":"def getPatientImages(patient_id):\n    counter=0\n    path = train_path + str(patient_id) +'/' \n    filePath = glob.glob(path +'*.dcm')\n    patient_images_num = len(filePath)\n    plt.figure(figsize = (24,15))\n    fig, axs = plt.subplots(2, 2, figsize=(20,12))\n    axs = axs.flatten()\n    for i, image in enumerate(filePath):\n        img = dicom.dcmread(image)\n        axs[i].set_title('Patient Image '+str(i+1))\n        axs[i].imshow(img.pixel_array, cmap=\"gray\")","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:39.877314Z","iopub.execute_input":"2023-01-08T12:02:39.877895Z","iopub.status.idle":"2023-01-08T12:02:39.888472Z","shell.execute_reply.started":"2023-01-08T12:02:39.877846Z","shell.execute_reply":"2023-01-08T12:02:39.88657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"getPatientImages(9989)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:42.394847Z","iopub.execute_input":"2023-01-08T12:02:42.395438Z","iopub.status.idle":"2023-01-08T12:02:45.647791Z","shell.execute_reply.started":"2023-01-08T12:02:42.395389Z","shell.execute_reply":"2023-01-08T12:02:45.64649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"getPatientImages(10011)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:50.885363Z","iopub.execute_input":"2023-01-08T12:02:50.885925Z","iopub.status.idle":"2023-01-08T12:02:55.276109Z","shell.execute_reply.started":"2023-01-08T12:02:50.88588Z","shell.execute_reply":"2023-01-08T12:02:55.274715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📌Patient Cancer Images","metadata":{}},{"cell_type":"code","source":"#plot patient cancer images \ndf[df['cancer']==1].head(100)\ngetPatientImages(14327)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:02:55.279223Z","iopub.execute_input":"2023-01-08T12:02:55.279735Z","iopub.status.idle":"2023-01-08T12:03:11.676844Z","shell.execute_reply.started":"2023-01-08T12:02:55.279689Z","shell.execute_reply":"2023-01-08T12:03:11.675527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📌Sample Of Data Normailzation in images\nVOI LUT stands for Value of Interest Look Up Table. This is a way of mapping 16 bit pixel values to 8 bit values. It's a different concept than regular normalization to 0-255 and allows for better visualization of the difference in tissue densities. ","metadata":{}},{"cell_type":"code","source":"def getPatientNormalizedImages(patient_id):\n    counter=0\n    path = train_path + str(patient_id) +'/' \n    filePath = glob.glob(path +'*.dcm')\n    patient_images_num = len(filePath)\n    plt.figure(figsize = (24,15))\n    fig, axs = plt.subplots(2, 2, figsize=(20,12))\n    axs = axs.flatten()\n    for i, image in enumerate(filePath):\n        img = dicom.dcmread(image)\n        img =img.pixel_array\n        img = apply_voi_lut(img, df, index=0)\n        axs[i].set_title('Patient Image '+str(i+1))\n        axs[i].imshow(img)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:03:11.678896Z","iopub.execute_input":"2023-01-08T12:03:11.679393Z","iopub.status.idle":"2023-01-08T12:03:11.689214Z","shell.execute_reply.started":"2023-01-08T12:03:11.679344Z","shell.execute_reply":"2023-01-08T12:03:11.688151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"getPatientNormalizedImages(14327)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T10:44:28.763665Z","iopub.execute_input":"2023-01-08T10:44:28.764134Z","iopub.status.idle":"2023-01-08T10:44:47.787902Z","shell.execute_reply.started":"2023-01-08T10:44:28.764087Z","shell.execute_reply":"2023-01-08T10:44:47.786777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"getPatientNormalizedImages(9989)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:03:11.691721Z","iopub.execute_input":"2023-01-08T12:03:11.692991Z","iopub.status.idle":"2023-01-08T12:03:14.741497Z","shell.execute_reply.started":"2023-01-08T12:03:11.692936Z","shell.execute_reply":"2023-01-08T12:03:14.740159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**EDA PART Completed , i will prepare for modiling soon.**","metadata":{}},{"cell_type":"markdown","source":"# 📌 Converting From DICOM to  PNG\n","metadata":{}},{"cell_type":"code","source":"SAVE_FOLDER = \"output/train/\"\nSIZE = 256\nEXTENSION = \"png\"\n\nos.makedirs(SAVE_FOLDER, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T10:44:51.7411Z","iopub.execute_input":"2023-01-08T10:44:51.741535Z","iopub.status.idle":"2023-01-08T10:44:51.747778Z","shell.execute_reply.started":"2023-01-08T10:44:51.741497Z","shell.execute_reply":"2023-01-08T10:44:51.746343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"has_cancer = df[df['cancer']==1]\nnormal     =  df[df['cancer']==0][0:3000]\nnew_df=     pd.concat([has_cancer,normal],axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:06:52.962045Z","iopub.execute_input":"2023-01-08T12:06:52.963682Z","iopub.status.idle":"2023-01-08T12:06:53.000352Z","shell.execute_reply.started":"2023-01-08T12:06:52.963622Z","shell.execute_reply":"2023-01-08T12:06:52.998822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# new_df['image']= df['image_path'].apply(image_name)    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(f, size=512, save_folder=\"\", extension=\"png\"):\n    #get patient id\n    patient = f.split('/')[-2]\n    #get image name\n    image = f.split('/')[-1][:-4]\n     #start read file \n    dicomfile = dicom.dcmread(f)\n    img = dicomfile.pixel_array\n    #resizing image \n    img = (img - img.min()) / (img.max() - img.min())\n  \n    if dicomfile.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n    img = cv2.resize(img, (size, size))\n\n    cv2.imwrite(save_folder + f\"{patient}_{image}.{extension}\", (img * 255).astype(np.uint8))","metadata":{"execution":{"iopub.status.busy":"2023-01-08T10:45:04.536505Z","iopub.execute_input":"2023-01-08T10:45:04.537032Z","iopub.status.idle":"2023-01-08T10:45:04.548222Z","shell.execute_reply.started":"2023-01-08T10:45:04.53699Z","shell.execute_reply":"2023-01-08T10:45:04.546491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from joblib import Parallel, delayed\nimport cv2\nfrom keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:07:06.356292Z","iopub.execute_input":"2023-01-08T12:07:06.356794Z","iopub.status.idle":"2023-01-08T12:07:12.732306Z","shell.execute_reply.started":"2023-01-08T12:07:06.356756Z","shell.execute_reply":"2023-01-08T12:07:12.731065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = Parallel(n_jobs=4)(\n    delayed(process)(uid, size=SIZE, save_folder=SAVE_FOLDER, extension=EXTENSION)\n    for uid in tqdm(new_df['image_path'])\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T10:45:18.356528Z","iopub.execute_input":"2023-01-08T10:45:18.357717Z","iopub.status.idle":"2023-01-08T11:16:12.624122Z","shell.execute_reply.started":"2023-01-08T10:45:18.357667Z","shell.execute_reply":"2023-01-08T11:16:12.62198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_FOLDER = \"output/test/\"\nSIZE = 256\nEXTENSION = \"png\"\n\nos.makedirs(SAVE_FOLDER, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T11:16:12.629967Z","iopub.execute_input":"2023-01-08T11:16:12.630697Z","iopub.status.idle":"2023-01-08T11:16:12.640063Z","shell.execute_reply.started":"2023-01-08T11:16:12.63063Z","shell.execute_reply":"2023-01-08T11:16:12.638781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = Parallel(n_jobs=4)(\n    delayed(process)(uid, size=SIZE, save_folder=SAVE_FOLDER, extension=EXTENSION)\n    for uid in tqdm(test['image_path'])\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T11:16:12.641662Z","iopub.execute_input":"2023-01-08T11:16:12.642929Z","iopub.status.idle":"2023-01-08T11:16:13.990796Z","shell.execute_reply.started":"2023-01-08T11:16:12.642879Z","shell.execute_reply":"2023-01-08T11:16:13.989683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shffle our df\nnew_df= new_df.sample(frac=1).reset_index(drop = True)\n#to be aadabtive with image genrator (from dataframe)\n# new_df[\"cancer\"] = np.where(new_df[\"cancer\"] == 1, 'infected', 'normal')","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:07:13.873567Z","iopub.execute_input":"2023-01-08T12:07:13.876393Z","iopub.status.idle":"2023-01-08T12:07:13.891153Z","shell.execute_reply.started":"2023-01-08T12:07:13.876303Z","shell.execute_reply":"2023-01-08T12:07:13.889841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_images = glob.glob('/kaggle/working/output/train/*.png')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:07:17.078444Z","iopub.execute_input":"2023-01-08T12:07:17.079044Z","iopub.status.idle":"2023-01-08T12:07:17.086683Z","shell.execute_reply.started":"2023-01-08T12:07:17.078992Z","shell.execute_reply":"2023-01-08T12:07:17.084795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_df['image_id']=new_df['image_id'].astype('object')","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:07:22.942371Z","iopub.execute_input":"2023-01-08T12:07:22.942824Z","iopub.status.idle":"2023-01-08T12:07:22.950651Z","shell.execute_reply.started":"2023-01-08T12:07:22.942787Z","shell.execute_reply":"2023-01-08T12:07:22.949025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pathes = new_df['new_path'].tolist()\nlabels =new_df['cancer'].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:07:24.515967Z","iopub.execute_input":"2023-01-08T12:07:24.516773Z","iopub.status.idle":"2023-01-08T12:07:24.523294Z","shell.execute_reply.started":"2023-01-08T12:07:24.51673Z","shell.execute_reply":"2023-01-08T12:07:24.52235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📌Loading Train Images","metadata":{}},{"cell_type":"code","source":"images =[]\nfor i,image  in enumerate(tqdm(pathes)):\n    image = cv2.imread(image)\n    image = cv2.resize(image,(256,256))\n    images.append(image)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:08:03.897008Z","iopub.execute_input":"2023-01-08T12:08:03.897488Z","iopub.status.idle":"2023-01-08T12:08:19.197422Z","shell.execute_reply.started":"2023-01-08T12:08:03.897447Z","shell.execute_reply":"2023-01-08T12:08:19.195892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📌 Train Images","metadata":{}},{"cell_type":"code","source":"    fig, axs = plt.subplots(3, 3, figsize=(20,12))\n    axs = axs.flatten()\n    for i, image in enumerate(images):\n        axs[i].set_title(get_Label(labels[i]))\n        axs[i].imshow(image)\n        axs[i].grid(False)\n        if i ==8 :\n            break","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:08:27.19294Z","iopub.execute_input":"2023-01-08T12:08:27.193411Z","iopub.status.idle":"2023-01-08T12:08:28.614627Z","shell.execute_reply.started":"2023-01-08T12:08:27.193372Z","shell.execute_reply":"2023-01-08T12:08:28.613403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(' Labels Visualization')\nsns.countplot(x=labels,palette='flare')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:08:33.421376Z","iopub.execute_input":"2023-01-08T12:08:33.421832Z","iopub.status.idle":"2023-01-08T12:08:33.637287Z","shell.execute_reply.started":"2023-01-08T12:08:33.421797Z","shell.execute_reply":"2023-01-08T12:08:33.635774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📌 Data Prepocessing\n\n1.   Data speliting\n2.   Data Normalization \n3.   Data Balansing\n\n","metadata":{}},{"cell_type":"code","source":"x_train,x_test,y_train,y_test  = train_test_split(images,labels ,random_state=42,shuffle=True,test_size=0.3)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:08:39.295569Z","iopub.execute_input":"2023-01-08T12:08:39.296137Z","iopub.status.idle":"2023-01-08T12:08:39.30594Z","shell.execute_reply.started":"2023-01-08T12:08:39.296093Z","shell.execute_reply":"2023-01-08T12:08:39.304774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data normalization\nx_train = np.asarray(x_train,np.float32)/255\nx_test  = np.asarray(x_test,np.float32) /255\ny_train = np.asarray(y_train)\ny_test  = np.asarray(y_test)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:08:42.026908Z","iopub.execute_input":"2023-01-08T12:08:42.028057Z","iopub.status.idle":"2023-01-08T12:08:43.422301Z","shell.execute_reply.started":"2023-01-08T12:08:42.028009Z","shell.execute_reply":"2023-01-08T12:08:43.421022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train Images shape is   : ',x_train.shape)\nprint('Train  Labels  shape is : ',y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:08:48.019407Z","iopub.execute_input":"2023-01-08T12:08:48.019915Z","iopub.status.idle":"2023-01-08T12:08:48.027225Z","shell.execute_reply.started":"2023-01-08T12:08:48.019873Z","shell.execute_reply":"2023-01-08T12:08:48.025828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Test Images shape is   : ',x_test.shape)\nprint('Test  Labels  shape is : ',y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:08:50.439093Z","iopub.execute_input":"2023-01-08T12:08:50.440438Z","iopub.status.idle":"2023-01-08T12:08:50.44684Z","shell.execute_reply.started":"2023-01-08T12:08:50.440392Z","shell.execute_reply":"2023-01-08T12:08:50.445481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Data  Balancing :👀 👀","metadata":{}},{"cell_type":"markdown","source":"The Imbalanced classification problem is what we face when there is a severe skew in the class distribution of our training data. Okay, the skew may not be extremely severe (it can vary), but the reason we identify imbalanced classification as a problem is because it can influence the performance on our Machine Learning algorithms.\nOne way the imbalance may affect our Machine Learning algorithm is when our algorithm completely ignores the minority class. The reason this is an issue is because the minority class is often the class that we are most interested in. For instance, when building a classifier to classify fraudulent and non-fraudulent transactions from various observations, the data is likely to have more non-fraudulent transactions than that of fraud — I mean think about it, it would be very worrying if we had an equal amount of fraudulent transactions as non-fraud.\n\n**Types** :-\n\nAn approach to combat this challenge is Random Sampling. There are two main ways to perform random resampling, both of which have there pros and cons:\n\n\n\n1.   Oversampling — Duplicating samples from the minority class\n\n2.  Undersampling — Deleting samples from the majority class.\n\n\n\n\n","metadata":{}},{"cell_type":"code","source":"# convert  Data  to 1D for  compatability oversampling method\nshape = 256*256*3\nx_train = x_train.reshape(x_train.shape[0],shape )\nx_test  = x_test.reshape(x_test.shape[0], shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:09:00.003008Z","iopub.execute_input":"2023-01-08T12:09:00.003472Z","iopub.status.idle":"2023-01-08T12:09:00.010414Z","shell.execute_reply.started":"2023-01-08T12:09:00.003434Z","shell.execute_reply":"2023-01-08T12:09:00.008958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('shape of new train data is :',x_train.shape)\nprint('shape of new Test data is :',x_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:09:03.265324Z","iopub.execute_input":"2023-01-08T12:09:03.265825Z","iopub.status.idle":"2023-01-08T12:09:03.272642Z","shell.execute_reply.started":"2023-01-08T12:09:03.265788Z","shell.execute_reply":"2023-01-08T12:09:03.271308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from imblearn.under_sampling import RandomOverSampler\nfrom imblearn.over_sampling import RandomOverSampler \n#call module\noversampler = RandomOverSampler(sampling_strategy='auto') \n#fit data to model\nx_train_fit, y_train_fit = oversampler.fit_resample(x_train, y_train)\nX_test_fit,  Y_test_fit   = oversampler.fit_resample(x_test, y_test)\n# return data to its original shape\nx_train_ = x_train_fit.reshape(x_train_fit.shape[0],256,256,3)\nx_test_  = X_test_fit.reshape(X_test_fit.shape[0], 256,256,3)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:12:03.341357Z","iopub.execute_input":"2023-01-08T12:12:03.341839Z","iopub.status.idle":"2023-01-08T12:12:07.528484Z","shell.execute_reply.started":"2023-01-08T12:12:03.3418Z","shell.execute_reply":"2023-01-08T12:12:07.527318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Visualization after Sampling Data :","metadata":{}},{"cell_type":"code","source":"plt.title('Train  Labels Visualization')\nsns.countplot(x=y_train_fit,palette='flare')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:12:11.760481Z","iopub.execute_input":"2023-01-08T12:12:11.760937Z","iopub.status.idle":"2023-01-08T12:12:11.984033Z","shell.execute_reply.started":"2023-01-08T12:12:11.760888Z","shell.execute_reply":"2023-01-08T12:12:11.982451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title('Test  Labels Visualization')\nsns.countplot(x=Y_test_fit,palette='flare')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:12:40.113584Z","iopub.execute_input":"2023-01-08T12:12:40.114115Z","iopub.status.idle":"2023-01-08T12:12:40.310042Z","shell.execute_reply.started":"2023-01-08T12:12:40.114073Z","shell.execute_reply":"2023-01-08T12:12:40.308646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train data shape =\",x_train_.shape)\nprint(\" Test data shape =\",x_test_.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:12:49.571824Z","iopub.execute_input":"2023-01-08T12:12:49.572911Z","iopub.status.idle":"2023-01-08T12:12:49.578436Z","shell.execute_reply.started":"2023-01-08T12:12:49.572868Z","shell.execute_reply":"2023-01-08T12:12:49.577303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 64\n#create image generator for images \nimage_gen = ImageDataGenerator(\n                               \n                                rotation_range=90,\n                                 width_shift_range=0.3,\n                                 height_shift_range=0.3,\n                                 shear_range=0.5,\n                                 zoom_range=0.3,\n                                 horizontal_flip=True,\n                                 vertical_flip=True,\n                                 fill_mode='reflect'\n)\ntrain = image_gen.flow(\n      x_train_,\n      y_train_fit,\n      shuffle=True, \n      batch_size=batch_size\n      )\ntest = image_gen.flow(\n      x_test_,\n      Y_test_fit,\n      shuffle=True, \n      batch_size=batch_size\n      )                                  \n        ","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:23:19.426411Z","iopub.execute_input":"2023-01-08T12:23:19.426892Z","iopub.status.idle":"2023-01-08T12:23:19.43595Z","shell.execute_reply.started":"2023-01-08T12:23:19.426852Z","shell.execute_reply":"2023-01-08T12:23:19.434584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn_model = Sequential()\ncnn_model.add(layers.Conv2D(256,(5,5), kernel_regularizer=l2(0.00005),padding ='Same',activation = 'relu',input_shape=(256,256,3)))\ncnn_model.add(BatchNormalization())\ncnn_model.add(layers.MaxPooling2D(2,2))\ncnn_model.add(layers.Conv2D(256,(3,3) ,kernel_regularizer=l2(0.00005),padding ='same',activation='relu'))\ncnn_model.add(BatchNormalization())\ncnn_model.add(Dropout(0.2))\ncnn_model.add(layers.MaxPooling2D(2,2))\ncnn_model.add(layers.Conv2D(128,(3,3) ,kernel_regularizer=l2(0.00005),padding ='same',activation='relu'))\ncnn_model.add(BatchNormalization())\ncnn_model.add(layers.MaxPooling2D(2,2)) \ncnn_model.add(layers.Conv2D(128,(3,3) ,kernel_regularizer=l2(0.00005),padding ='same',activation='relu'))\ncnn_model.add(BatchNormalization())\ncnn_model.add(Dropout(0.2))\ncnn_model.add(layers.MaxPooling2D(2,2)) ","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:37:26.313538Z","iopub.execute_input":"2023-01-08T12:37:26.314026Z","iopub.status.idle":"2023-01-08T12:37:26.457898Z","shell.execute_reply.started":"2023-01-08T12:37:26.313988Z","shell.execute_reply":"2023-01-08T12:37:26.456704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn_model.add(layers.Flatten())\ncnn_model.add(layers.Dense(128, activation='relu'))\ncnn_model.add(BatchNormalization())\ncnn_model.add(Dropout(0.1))\ncnn_model.add(layers.Dense(1, activation ='sigmoid'))\ncnn_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:37:31.408163Z","iopub.execute_input":"2023-01-08T12:37:31.408681Z","iopub.status.idle":"2023-01-08T12:37:31.484496Z","shell.execute_reply.started":"2023-01-08T12:37:31.408633Z","shell.execute_reply":"2023-01-08T12:37:31.483214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn_model.compile(optimizer='adam',loss='binary_crossentropy',metrics=['accuracy'],)","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:37:33.193865Z","iopub.execute_input":"2023-01-08T12:37:33.19438Z","iopub.status.idle":"2023-01-08T12:37:33.205604Z","shell.execute_reply.started":"2023-01-08T12:37:33.194341Z","shell.execute_reply":"2023-01-08T12:37:33.204594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"  **1-Defining Callbacks**\n\n*   A callback is an object that can perform actions at various stages of training (e.g. at the start or end of an epoch, before or after a single batch, etc)\n\n\n**2-Reduce Learning Rate on Plateau**\n*   Is used to reduce the learning rate when a metric has stopped improving.\n\n","metadata":{}},{"cell_type":"code","source":"early = EarlyStopping(monitor=\"loss\", mode=\"min\",min_delta = 0,\n                          patience = 10,\n                          verbose = 1,\n                          restore_best_weights = True)\nlearning_rate_reduction = ReduceLROnPlateau(monitor='loss', patience = 2, verbose=1,factor=0.3, min_lr=0.000001)\ncallbacks_list = [ early, learning_rate_reduction]","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:37:34.585387Z","iopub.execute_input":"2023-01-08T12:37:34.585833Z","iopub.status.idle":"2023-01-08T12:37:34.594357Z","shell.execute_reply.started":"2023-01-08T12:37:34.585799Z","shell.execute_reply":"2023-01-08T12:37:34.593218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training model\nn_training_samples = len(train)\nn_validation_samples = len(test)\nhistory = cnn_model.fit(\n    train,\n    epochs=50,\n    validation_data=test,\n    validation_steps=n_validation_samples//batch_size,\n    # steps_per_epoch =n_training_samples//batch_size,\n    shuffle = True,\n    callbacks=callbacks_list\n    )","metadata":{"execution":{"iopub.status.busy":"2023-01-08T12:37:36.726501Z","iopub.execute_input":"2023-01-08T12:37:36.726955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# i will resume my work as i face issue of Memory ,if you have asolution please raise it to me ","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import shutil\n# shutil.rmtree(\"/kaggle/working/output/\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('/kaggle/working/output')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len('/kaggle/working/output/test')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x=glob.glob('/kaggle/working/output/test/*.png')\nlen(x)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_df['new_path'][0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}