{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nimport OpenSSL\n# import timm\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os \nimport pydicom\nimport seaborn as sns\nimport tensorflow as tf\nimport tensorflow_io as tfio\nfrom PIL import Image\nimport cv2\nfrom tensorflow.keras.models import load_model\nimport csv\nfrom tensorflow.keras.layers import Input, Concatenate, Dense, Flatten, Conv2D, MaxPooling2D\nfrom tensorflow.keras.models import Model\nfrom sklearn.preprocessing import StandardScaler\nfrom keras import layers","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-22T19:20:43.582216Z","iopub.execute_input":"2023-03-22T19:20:43.582617Z","iopub.status.idle":"2023-03-22T19:20:53.004582Z","shell.execute_reply.started":"2023-03-22T19:20:43.58258Z","shell.execute_reply":"2023-03-22T19:20:53.003181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath_mammography = \"/kaggle/input/rsna-breast-cancer-detection/train.csv\"\n\ndata_mammography = pd.read_csv(filepath_mammography)","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:53.006837Z","iopub.execute_input":"2023-03-22T19:20:53.007557Z","iopub.status.idle":"2023-03-22T19:20:53.132021Z","shell.execute_reply.started":"2023-03-22T19:20:53.007518Z","shell.execute_reply":"2023-03-22T19:20:53.131004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Find the correlation of the dataset\n\ncorr= data_mammography.corr()\n\n#Plot heatmap\nplt.figure(figsize=(15,5))\nsns.heatmap(corr, annot=True, vmin=-1.0, cmap ='mako')\nplt.title('Correlation')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:53.133915Z","iopub.execute_input":"2023-03-22T19:20:53.134285Z","iopub.status.idle":"2023-03-22T19:20:54.282396Z","shell.execute_reply.started":"2023-03-22T19:20:53.134249Z","shell.execute_reply":"2023-03-22T19:20:54.281078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So i can see that i have \"cancer\", \"biopsy\" and \"invasive\" that have a lot of similarity, so i will just keep the column cancer.","metadata":{}},{"cell_type":"code","source":"data_mammography = data_mammography.drop(columns=['biopsy', 'invasive'])\ndata_mammography.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.284814Z","iopub.execute_input":"2023-03-22T19:20:54.285169Z","iopub.status.idle":"2023-03-22T19:20:54.321528Z","shell.execute_reply.started":"2023-03-22T19:20:54.285136Z","shell.execute_reply":"2023-03-22T19:20:54.320172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## let's see missing Value ","metadata":{}},{"cell_type":"code","source":"for column in data_mammography.columns :\n    print(column,\":\",data_mammography[column].isnull().sum(), \" / \" , len(data_mammography[column]))\n    print(data_mammography[column].isnull().sum()/len(data_mammography[column])*100,\"%\")\n    print(\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.322985Z","iopub.execute_input":"2023-03-22T19:20:54.323418Z","iopub.status.idle":"2023-03-22T19:20:54.362022Z","shell.execute_reply.started":"2023-03-22T19:20:54.323375Z","shell.execute_reply":"2023-03-22T19:20:54.360649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that there is 3 columns with Missing values, density is an important column so i will replace each missing value by the meaning of it, i will make the same for age but BIRADS not very useful for my model so i just drop it ","metadata":{}},{"cell_type":"code","source":"# data_mammography.groupby(['BIRADS','cancer']).size().sort_values(ascending=False)\ndata_mammography = data_mammography.drop(columns=['BIRADS'])","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.363793Z","iopub.execute_input":"2023-03-22T19:20:54.364257Z","iopub.status.idle":"2023-03-22T19:20:54.375511Z","shell.execute_reply.started":"2023-03-22T19:20:54.364221Z","shell.execute_reply":"2023-03-22T19:20:54.374449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LETS REPLACE MISSING VALUES","metadata":{}},{"cell_type":"code","source":"data_mammography[\"age\"] = data_mammography[\"age\"].fillna(data_mammography[\"age\"].mean())","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.376768Z","iopub.execute_input":"2023-03-22T19:20:54.37756Z","iopub.status.idle":"2023-03-22T19:20:54.388474Z","shell.execute_reply.started":"2023-03-22T19:20:54.37752Z","shell.execute_reply":"2023-03-22T19:20:54.387545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Before replace missing values we need to reattribute value of \"density\" because its a categorical objet ","metadata":{}},{"cell_type":"code","source":"data_mammography[\"density\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.38993Z","iopub.execute_input":"2023-03-22T19:20:54.390313Z","iopub.status.idle":"2023-03-22T19:20:54.406881Z","shell.execute_reply.started":"2023-03-22T19:20:54.390266Z","shell.execute_reply":"2023-03-22T19:20:54.405265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\n# Apply ordinal encoder to each column with categorical data\nordinal_encoder = OrdinalEncoder()\ndata_mammography[[\"density\"]] = ordinal_encoder.fit_transform(data_mammography[[\"density\"]])","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.408417Z","iopub.execute_input":"2023-03-22T19:20:54.408763Z","iopub.status.idle":"2023-03-22T19:20:54.435831Z","shell.execute_reply.started":"2023-03-22T19:20:54.408731Z","shell.execute_reply":"2023-03-22T19:20:54.434779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_mammography.density.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.441372Z","iopub.execute_input":"2023-03-22T19:20:54.441726Z","iopub.status.idle":"2023-03-22T19:20:54.451786Z","shell.execute_reply.started":"2023-03-22T19:20:54.441693Z","shell.execute_reply":"2023-03-22T19:20:54.45085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we have density with a float value so now i will make a model that recognize the density of a breast with picture of patients","metadata":{}},{"cell_type":"code","source":"model_X_Value = data_mammography.copy()\n\nmodel_X_Value = model_X_Value[pd.notnull(model_X_Value.density)]\n\nmodel_Y = model_X_Value[\"density\"]","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.452695Z","iopub.execute_input":"2023-03-22T19:20:54.453046Z","iopub.status.idle":"2023-03-22T19:20:54.475884Z","shell.execute_reply.started":"2023-03-22T19:20:54.453013Z","shell.execute_reply":"2023-03-22T19:20:54.474777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_X_Value.site_id.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.47754Z","iopub.execute_input":"2023-03-22T19:20:54.477874Z","iopub.status.idle":"2023-03-22T19:20:54.486619Z","shell.execute_reply.started":"2023-03-22T19:20:54.477841Z","shell.execute_reply":"2023-03-22T19:20:54.485405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# site_id has only one value so we can drop this columns ","metadata":{}},{"cell_type":"code","source":"model_X_Value.laterality.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.488795Z","iopub.execute_input":"2023-03-22T19:20:54.489249Z","iopub.status.idle":"2023-03-22T19:20:54.49975Z","shell.execute_reply.started":"2023-03-22T19:20:54.489199Z","shell.execute_reply":"2023-03-22T19:20:54.498702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_X_Value.machine_id.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.500812Z","iopub.execute_input":"2023-03-22T19:20:54.501226Z","iopub.status.idle":"2023-03-22T19:20:54.511606Z","shell.execute_reply.started":"2023-03-22T19:20:54.501182Z","shell.execute_reply":"2023-03-22T19:20:54.510567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_X_Value.view.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.513204Z","iopub.execute_input":"2023-03-22T19:20:54.513549Z","iopub.status.idle":"2023-03-22T19:20:54.527082Z","shell.execute_reply.started":"2023-03-22T19:20:54.513514Z","shell.execute_reply":"2023-03-22T19:20:54.5258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We can see that we have 2 values for this columns that is redondant so we can drop the other","metadata":{}},{"cell_type":"code","source":"model_Y = np.array(model_Y)","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.528679Z","iopub.execute_input":"2023-03-22T19:20:54.529175Z","iopub.status.idle":"2023-03-22T19:20:54.536588Z","shell.execute_reply.started":"2023-03-22T19:20:54.529125Z","shell.execute_reply":"2023-03-22T19:20:54.535137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_Y.size","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.538181Z","iopub.execute_input":"2023-03-22T19:20:54.538685Z","iopub.status.idle":"2023-03-22T19:20:54.548702Z","shell.execute_reply.started":"2023-03-22T19:20:54.538636Z","shell.execute_reply":"2023-03-22T19:20:54.547405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# i drop some column that is useless for what we need to find \n\ncolumnsDrop = [\"difficult_negative_case\",\"density\",\"site_id\",\"laterality\"]\nmodel_X_Value = model_X_Value.drop(columns=columnsDrop)\nmodel_X_Value = model_X_Value.loc[(model_X_Value.view.isin(['CC','MLO']))]\nmodel_X_Value[[\"view\"]] = ordinal_encoder.fit_transform(model_X_Value[[\"view\"]])","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.551469Z","iopub.execute_input":"2023-03-22T19:20:54.552244Z","iopub.status.idle":"2023-03-22T19:20:54.576766Z","shell.execute_reply.started":"2023-03-22T19:20:54.552205Z","shell.execute_reply":"2023-03-22T19:20:54.57528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"size : \",model_X_Value.size)\n\nmodel_X_Value.view.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.578187Z","iopub.execute_input":"2023-03-22T19:20:54.578555Z","iopub.status.idle":"2023-03-22T19:20:54.589226Z","shell.execute_reply.started":"2023-03-22T19:20:54.57852Z","shell.execute_reply":"2023-03-22T19:20:54.588121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## now MLO view's is 1.0 and CC is 0.0","metadata":{}},{"cell_type":"code","source":"model_X_Value.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.590219Z","iopub.execute_input":"2023-03-22T19:20:54.590835Z","iopub.status.idle":"2023-03-22T19:20:54.606127Z","shell.execute_reply.started":"2023-03-22T19:20:54.590799Z","shell.execute_reply":"2023-03-22T19:20:54.604927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We keep \"patient_id and images_id to get image_path for making our model object detection","metadata":{}},{"cell_type":"markdown","source":"## lets make a function for convert image DCM to PNG ","metadata":{}},{"cell_type":"code","source":"imagesDCMPath = []\npath = \"/kaggle/input/rsna-breast-cancer-detection/train_images/\"\nfor index, row in model_X_Value.iterrows():\n    imagesDCMPath.append(path+str(int(row[\"patient_id\"]))+\"/\"+str(int(row[\"image_id\"]))+\".dcm\")","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:54.607547Z","iopub.execute_input":"2023-03-22T19:20:54.608504Z","iopub.status.idle":"2023-03-22T19:20:56.223941Z","shell.execute_reply.started":"2023-03-22T19:20:54.608466Z","shell.execute_reply":"2023-03-22T19:20:56.222494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resizing_image(numpyarray) :\n    image = cv2.resize(numpyarray, (64,64), interpolation=cv2.INTER_AREA)\n    return image","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.225579Z","iopub.execute_input":"2023-03-22T19:20:56.22599Z","iopub.status.idle":"2023-03-22T19:20:56.23215Z","shell.execute_reply.started":"2023-03-22T19:20:56.225949Z","shell.execute_reply":"2023-03-22T19:20:56.230746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_image(filepath) :\n    image_bytes = tf.io.read_file(filepath)\n    pixel_data = tfio.image.decode_dicom_image(\n        image_bytes,\n        dtype=tf.uint16)\n    pixel_data_nparray = pixel_data.numpy()\n\n    pixel_data_nparray.shape = (len(pixel_data_nparray[0]), len(pixel_data_nparray[0][1]))\n    return resizing_image(pixel_data_nparray)","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.234113Z","iopub.execute_input":"2023-03-22T19:20:56.234633Z","iopub.status.idle":"2023-03-22T19:20:56.245652Z","shell.execute_reply.started":"2023-03-22T19:20:56.234584Z","shell.execute_reply":"2023-03-22T19:20:56.244601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's analyse the images","metadata":{}},{"cell_type":"code","source":"# time = 0\n# path = \"/kaggle/input/rsna-breast-cancer-detection/train_images/\"\n# for index, row in model_X_Value.iterrows():\n#     print(\"image n°\",time)\n#     print(\"image_id : \" + str(int(row[\"image_id\"])))\n#     print(\"patient_id : \" + str(int(row[\"patient_id\"])))\n#     typeimg = str(int(row[\"view\"]))\n#     if typeimg == \"1\" :\n#         typeimg = \"MLO\"\n#     else :\n#         typeimg = \"CC\"\n#     print(\"type of image : \" + typeimg)\n#     print(\"id_machine : \" + str(int(row[\"machine_id\"])))\n#     pltimg = convert_image(path+str(int(row[\"patient_id\"]))+\"/\"+str(int(row[\"image_id\"]))+\".dcm\")\n#     plt.imshow(pltimg,'gray')\n#     plt.show()\n#     time+=1\n#     if time == 5 :break","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.247148Z","iopub.execute_input":"2023-03-22T19:20:56.247702Z","iopub.status.idle":"2023-03-22T19:20:56.25619Z","shell.execute_reply.started":"2023-03-22T19:20:56.24762Z","shell.execute_reply":"2023-03-22T19:20:56.25515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# time = 0\n# liste_numpy = []\n# path = \"/kaggle/input/rsna-breast-cancer-detection/train_images/\"\n# for index, row in model_X_Value.iterrows():\n#     liste_numpy.append(convert_image(path+str(int(row[\"patient_id\"]))+\"/\"+str(int(row[\"image_id\"]))+\".dcm\"))\n#     time+=1\n#     if time == 50 :\n#         fig, axes = plt.subplots(nrows=5, ncols=10)\n\n#         # Parcourir  chaque sous-graphique et y afficher le tableau numpy correspondant\n#         for i, ax in enumerate(axes.flat):\n#             ax.imshow(liste_numpy[i], cmap='gray')\n#             ax.set_xticks([])\n#             ax.set_yticks([])\n\n#         # Afficher la figure\n#         plt.show()\n#         break","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-03-22T19:20:56.257774Z","iopub.execute_input":"2023-03-22T19:20:56.258209Z","iopub.status.idle":"2023-03-22T19:20:56.270019Z","shell.execute_reply.started":"2023-03-22T19:20:56.25817Z","shell.execute_reply":"2023-03-22T19:20:56.268833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# npArrayDataImgModel = []\n# for index in range(1400) :\n#     npArrayDataImgModel.append(convert_image(imagesDCMPath[index]))\n#     print(index)","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2023-03-22T19:20:56.271192Z","iopub.execute_input":"2023-03-22T19:20:56.271557Z","iopub.status.idle":"2023-03-22T19:20:56.281145Z","shell.execute_reply.started":"2023-03-22T19:20:56.271521Z","shell.execute_reply":"2023-03-22T19:20:56.279872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# img_width = 64\n# img_height = 64\n# model_Y = model_Y[:len(npArrayDataImgModel)]\n# # Diviser l'ensemble de données en un ensemble d'entraînement et un ensemble de test\n# split_idx = int(0.8 * len(npArrayDataImgModel))\n# X_train, y_train = np.array(npArrayDataImgModel[:split_idx]), model_Y[:split_idx]\n# X_test, y_test = np.array(npArrayDataImgModel[split_idx:]), model_Y[split_idx:]","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.282363Z","iopub.execute_input":"2023-03-22T19:20:56.283593Z","iopub.status.idle":"2023-03-22T19:20:56.290113Z","shell.execute_reply.started":"2023-03-22T19:20:56.283552Z","shell.execute_reply":"2023-03-22T19:20:56.288959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Créer le modèle feedforward","metadata":{}},{"cell_type":"code","source":"# model = tf.keras.Sequential([\n#     tf.keras.layers.InputLayer(input_shape=(64, 64, 1)),\n#     tf.keras.layers.Conv2D(filters=16, kernel_size=3, activation='relu'),\n#     tf.keras.layers.MaxPooling2D(),\n#     tf.keras.layers.Conv2D(filters=32, kernel_size=3, activation='relu'),\n#     tf.keras.layers.MaxPooling2D(),\n#     tf.keras.layers.Flatten(),\n#     tf.keras.layers.Dense(64, activation='relu'),\n#     tf.keras.layers.Dropout(0.5),\n#     tf.keras.layers.Dense(4, activation='softmax')\n# ])","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.298949Z","iopub.execute_input":"2023-03-22T19:20:56.299395Z","iopub.status.idle":"2023-03-22T19:20:56.305053Z","shell.execute_reply.started":"2023-03-22T19:20:56.299331Z","shell.execute_reply":"2023-03-22T19:20:56.303517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Compiler le modèle","metadata":{}},{"cell_type":"code","source":"# model.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.306965Z","iopub.execute_input":"2023-03-22T19:20:56.307449Z","iopub.status.idle":"2023-03-22T19:20:56.31783Z","shell.execute_reply.started":"2023-03-22T19:20:56.307401Z","shell.execute_reply":"2023-03-22T19:20:56.316614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Entraîner le modèle sur l'ensemble d'entraînement\n","metadata":{}},{"cell_type":"code","source":"# model.fit(X_train, y_train, epochs=25,batch_size=2)","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.318991Z","iopub.execute_input":"2023-03-22T19:20:56.319573Z","iopub.status.idle":"2023-03-22T19:20:56.325448Z","shell.execute_reply.started":"2023-03-22T19:20:56.319523Z","shell.execute_reply":"2023-03-22T19:20:56.324428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Évaluer les performances du modèle sur l'ensemble de test\n","metadata":{}},{"cell_type":"code","source":"# val_loss, val_acc = model.evaluate(X_test, y_test)\n# print(\"Validation accuracy:\", val_acc)","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.326845Z","iopub.execute_input":"2023-03-22T19:20:56.328074Z","iopub.status.idle":"2023-03-22T19:20:56.333747Z","shell.execute_reply.started":"2023-03-22T19:20:56.328035Z","shell.execute_reply":"2023-03-22T19:20:56.332792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.save(\"model_density.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.334964Z","iopub.execute_input":"2023-03-22T19:20:56.336463Z","iopub.status.idle":"2023-03-22T19:20:56.342263Z","shell.execute_reply.started":"2023-03-22T19:20:56.336415Z","shell.execute_reply":"2023-03-22T19:20:56.341401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict Density of breast","metadata":{}},{"cell_type":"markdown","source":"# Charger le modèle\n","metadata":{}},{"cell_type":"code","source":"# model = tf.keras.models.load_model(\"/kaggle/input/modeldensity/model_density.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.34376Z","iopub.execute_input":"2023-03-22T19:20:56.344099Z","iopub.status.idle":"2023-03-22T19:20:56.350392Z","shell.execute_reply.started":"2023-03-22T19:20:56.344065Z","shell.execute_reply":"2023-03-22T19:20:56.349466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predictDensity(filepath) :\n    # Faire la prédiction\n    img = convert_image(filepath)\n    img = img.reshape(-1, 64, 64, 1)\n    predictions = model.predict(img)\n\n    # Trouver la catégorie prédite\n    predicted_class = np.argmax(predictions)\n\n    # Afficher la catégorie prédite\n    return predicted_class","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.351795Z","iopub.execute_input":"2023-03-22T19:20:56.352409Z","iopub.status.idle":"2023-03-22T19:20:56.359191Z","shell.execute_reply.started":"2023-03-22T19:20:56.352367Z","shell.execute_reply":"2023-03-22T19:20:56.358304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# imageToBeTraited = []\n# DataFrameNullDensity = pd.read_csv(\"/kaggle/input/changingdensity6/changingDensityple6.csv\")\n# DataFrameNullDensity = DataFrameNullDensity[pd.isnull(DataFrameNullDensity.density)]\n# path = \"/kaggle/input/rsna-breast-cancer-detection/train_images/\"\n# for index, row in DataFrameNullDensity.iterrows():\n#     imageToBeTraited.append(path+str(int(row[\"patient_id\"]))+\"/\"+str(int(row[\"image_id\"]))+\".dcm\")","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.3612Z","iopub.execute_input":"2023-03-22T19:20:56.362242Z","iopub.status.idle":"2023-03-22T19:20:56.368285Z","shell.execute_reply.started":"2023-03-22T19:20:56.362207Z","shell.execute_reply":"2023-03-22T19:20:56.367411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DataFrameNullDensity","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.369517Z","iopub.execute_input":"2023-03-22T19:20:56.37004Z","iopub.status.idle":"2023-03-22T19:20:56.378252Z","shell.execute_reply.started":"2023-03-22T19:20:56.370006Z","shell.execute_reply":"2023-03-22T19:20:56.377289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predictShow = []\n# count = 0\n# for image in imageToBeTraited :\n#     predictShow.append(predictDensity(image))\n#     count+=1\n#     print(count)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-03-22T19:20:56.379382Z","iopub.execute_input":"2023-03-22T19:20:56.380107Z","iopub.status.idle":"2023-03-22T19:20:56.388655Z","shell.execute_reply.started":"2023-03-22T19:20:56.380074Z","shell.execute_reply":"2023-03-22T19:20:56.387441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"0 : \" + str(predictShow.count(0)))\n# print(\"1 : \" + str(predictShow.count(1)))\n# print(\"2 : \" + str(predictShow.count(2)))\n# print(\"3 : \" + str(predictShow.count(3)))\n# print(len(predictShow))","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.390507Z","iopub.execute_input":"2023-03-22T19:20:56.391191Z","iopub.status.idle":"2023-03-22T19:20:56.397518Z","shell.execute_reply.started":"2023-03-22T19:20:56.391153Z","shell.execute_reply":"2023-03-22T19:20:56.39617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DataFrameNullDensity[\"density\"][:len(predictShow)] = predictShow","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.398728Z","iopub.execute_input":"2023-03-22T19:20:56.399402Z","iopub.status.idle":"2023-03-22T19:20:56.407348Z","shell.execute_reply.started":"2023-03-22T19:20:56.399358Z","shell.execute_reply":"2023-03-22T19:20:56.406289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DataFrameNullDensity","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.408853Z","iopub.execute_input":"2023-03-22T19:20:56.409563Z","iopub.status.idle":"2023-03-22T19:20:56.429812Z","shell.execute_reply.started":"2023-03-22T19:20:56.409526Z","shell.execute_reply":"2023-03-22T19:20:56.42861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# know lets make the same for cancer detection","metadata":{}},{"cell_type":"code","source":"DensityChange = pd.read_csv(\"/kaggle/input/changingdensitydata/ChangingDensity.csv\") # its the csv replace by the value of the densityModel","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.4317Z","iopub.execute_input":"2023-03-22T19:20:56.432693Z","iopub.status.idle":"2023-03-22T19:20:56.503452Z","shell.execute_reply.started":"2023-03-22T19:20:56.432643Z","shell.execute_reply":"2023-03-22T19:20:56.502078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DensityChange.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.506166Z","iopub.execute_input":"2023-03-22T19:20:56.506608Z","iopub.status.idle":"2023-03-22T19:20:56.52447Z","shell.execute_reply.started":"2023-03-22T19:20:56.50657Z","shell.execute_reply":"2023-03-22T19:20:56.523024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FinalDataSet = DensityChange.append(data_mammography.copy()[pd.notnull(data_mammography.copy().density)],ignore_index = True)","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.526726Z","iopub.execute_input":"2023-03-22T19:20:56.527091Z","iopub.status.idle":"2023-03-22T19:20:56.548793Z","shell.execute_reply.started":"2023-03-22T19:20:56.527054Z","shell.execute_reply":"2023-03-22T19:20:56.54755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FinalDataSet.density.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.550128Z","iopub.execute_input":"2023-03-22T19:20:56.550592Z","iopub.status.idle":"2023-03-22T19:20:56.561873Z","shell.execute_reply.started":"2023-03-22T19:20:56.550556Z","shell.execute_reply":"2023-03-22T19:20:56.560727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calcul Cancer ","metadata":{}},{"cell_type":"code","source":"# input layer for the image\nimage_input = Input(shape=(64, 64, 1), name='input_img')\ninput_df = Input((5,), name='input_df')\n\nx = Conv2D(16, (3,3), activation='relu', padding='same')(image_input)\nx = MaxPooling2D(pool_size=(2,2))(x)\nx = Conv2D(32, (3,3), activation='relu', padding='same')(x)\nx = MaxPooling2D(pool_size=(2,2))(x)\n\nflat_layer = Flatten()(x)\n\n# Concatenate the convolutional features and the vector input\nconcat_layer= Concatenate()([input_df, flat_layer])\nconcat_layer = layers.Dense(64, activation='relu')(concat_layer)\noutput = Dense(1, activation='sigmoid')(concat_layer)\n\n# define a model with a list of two inputs\nmodel = Model(inputs=[image_input, input_df], outputs=output, name='cancer_classifier')","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.563912Z","iopub.execute_input":"2023-03-22T19:20:56.564297Z","iopub.status.idle":"2023-03-22T19:20:56.789252Z","shell.execute_reply.started":"2023-03-22T19:20:56.564262Z","shell.execute_reply":"2023-03-22T19:20:56.788209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compilation du modèle\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.790822Z","iopub.execute_input":"2023-03-22T19:20:56.791196Z","iopub.status.idle":"2023-03-22T19:20:56.813641Z","shell.execute_reply.started":"2023-03-22T19:20:56.79116Z","shell.execute_reply":"2023-03-22T19:20:56.812512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's prepare the data ","metadata":{}},{"cell_type":"code","source":"idPatient = FinalDataSet.patient_id\nidImage = FinalDataSet.image_id\nX = FinalDataSet.drop(columns=['site_id', 'image_id',\"patient_id\",\"laterality\",\"machine_id\"]) # on enlève les variables inutiles\nX[\"age\"].fillna(X[\"age\"].mean()) # on remplie les valeurs manquantes de la colonne age\nX = X.loc[(X.view.isin(['CC','MLO']))] # on ne garde que les points de vue \"CC\" et \"MLO\"\nX[[\"view\"]] = ordinal_encoder.fit_transform(X[[\"view\"]]) # on change les points de vu par des nombre pour faire un model correct\nX = X.astype(float) # on mets toutes les valeurs en float notament pour la colonne \"difficult_negative_case\" qui étais booléenne\nX = X.reset_index(drop=True)\ny = X.cancer.values #y est la variable que l'on doit trouver\nX = X.drop(columns=[\"cancer\"])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-03-22T19:20:56.815114Z","iopub.execute_input":"2023-03-22T19:20:56.815595Z","iopub.status.idle":"2023-03-22T19:20:56.866064Z","shell.execute_reply.started":"2023-03-22T19:20:56.815544Z","shell.execute_reply":"2023-03-22T19:20:56.864766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:20:56.867508Z","iopub.execute_input":"2023-03-22T19:20:56.868501Z","iopub.status.idle":"2023-03-22T19:20:56.903099Z","shell.execute_reply.started":"2023-03-22T19:20:56.868458Z","shell.execute_reply":"2023-03-22T19:20:56.90161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"npArrayImg = []\npath = \"/kaggle/input/rsna-breast-cancer-detection/train_images/\"\nfor index in range(100): # ici je ne prend que les 100 premiers élément de X pour enregistrer mon travail mais il y plus de 52 000 éléments dans X \n    xDataFrame = X.loc[index]\n    xImage = convert_image(path+str(idPatient[index])+\"/\"+str(idImage[index])+\".dcm\")\n    npArrayImg.append([xImage, xDataFrame.to_numpy()])\n    print(index)","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Empiler les images et les vecteurs de données en une sexdule matrice\nimages = np.stack([x[0] for x in npArrayImg], axis=0)\ndata = np.stack([x[1] for x in npArrayImg], axis=0)\n\nXModel = [images, data]\n\nmodel.fit(XModel ,y, epochs=25 ,validation_split=0.2,batch_size=2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save(\"modelCancerDetection.h5\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prédiction sur de nouvelles données\npredictData = []\nxDataFrame = X.loc[len(X)-1]\nxImage = convert_image(path+str(idPatient[len(X)-1])+\"/\"+str(idImage[len(X)-1])+\".dcm\")\npredictData.append([xImage, xDataFrame.to_numpy()])\n\nimage = np.stack([x[0] for x in predictData], axis=0)\ndata = np.stack([x[1] for x in predictData], axis=0)\n\ny_pred = model.predict([image,data])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Probabilité de cancer :\", y_pred[0][0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y[len(X)-1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save(\"modelCancerDetection2.h5\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}