{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg\n","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:31:04.335893Z","iopub.execute_input":"2023-08-26T09:31:04.336299Z","iopub.status.idle":"2023-08-26T09:31:21.236657Z","shell.execute_reply.started":"2023-08-26T09:31:04.336262Z","shell.execute_reply":"2023-08-26T09:31:21.235404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" !pip install numpy","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:31:26.494171Z","iopub.execute_input":"2023-08-26T09:31:26.494896Z","iopub.status.idle":"2023-08-26T09:31:37.718496Z","shell.execute_reply.started":"2023-08-26T09:31:26.494861Z","shell.execute_reply":"2023-08-26T09:31:37.717379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport glob\nimport gdcm\nimport pydicom\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:31:41.602641Z","iopub.execute_input":"2023-08-26T09:31:41.60322Z","iopub.status.idle":"2023-08-26T09:31:43.068023Z","shell.execute_reply.started":"2023-08-26T09:31:41.603181Z","shell.execute_reply":"2023-08-26T09:31:43.067047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:31:47.666874Z","iopub.execute_input":"2023-08-26T09:31:47.667326Z","iopub.status.idle":"2023-08-26T09:31:59.341083Z","shell.execute_reply.started":"2023-08-26T09:31:47.667288Z","shell.execute_reply":"2023-08-26T09:31:59.339897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow.keras.backend as K\nimport matplotlib.image as mpimg\npd.options.display.max_columns = 50","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:32:28.233557Z","iopub.execute_input":"2023-08-26T09:32:28.234316Z","iopub.status.idle":"2023-08-26T09:32:28.239704Z","shell.execute_reply.started":"2023-08-26T09:32:28.234281Z","shell.execute_reply":"2023-08-26T09:32:28.238608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(f, size=512, save_folder=\"\", extension=\"png\"):\n    patient = f.split('/')[-2]\n    image = f.split('/')[-1][:-4]\n\n    dicom = pydicom.dcmread(f)\n    img = dicom.pixel_array\n\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n\n    img = cv2.resize(img, (size, size))\n    cv2.imwrite(save_folder + f\"{patient}_{image}.{extension}\", (img * 255).astype(np.uint8))\n","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:32:33.930972Z","iopub.execute_input":"2023-08-26T09:32:33.931377Z","iopub.status.idle":"2023-08-26T09:32:33.938688Z","shell.execute_reply.started":"2023-08-26T09:32:33.931345Z","shell.execute_reply":"2023-08-26T09:32:33.937696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_pngs_from_dcms(dcm_path,SAVE_FOLDER,SIZE,EXTENSION):\n    train_images = glob.glob(dcm_path)\n    len(train_images)  # 54706\n    print('...........Images loaded')\n    \n    os.makedirs(SAVE_FOLDER, exist_ok=True)\n    # an empty directory called train_output in kaggle/working path is created\n    print('...........New folder created')\n    \n    _ = Parallel(n_jobs=4)(\n    delayed(process)(uid, size=SIZE, save_folder=SAVE_FOLDER, extension=EXTENSION)\n    for uid in tqdm(train_images)\n    )\n    \n    print('............Finished')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:32:38.331139Z","iopub.execute_input":"2023-08-26T09:32:38.331542Z","iopub.status.idle":"2023-08-26T09:32:38.338779Z","shell.execute_reply.started":"2023-08-26T09:32:38.331512Z","shell.execute_reply":"2023-08-26T09:32:38.337532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dcm_path = '/kaggle/input/rsna-breast-cancer-detection/test_images/*/*.dcm'\nSAVE_FOLDER = \"test_output/\"\nSIZE = 512\nEXTENSION = \"png\"\ncreate_pngs_from_dcms(dcm_path,SAVE_FOLDER,SIZE,EXTENSION)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:32:43.748366Z","iopub.execute_input":"2023-08-26T09:32:43.749089Z","iopub.status.idle":"2023-08-26T09:32:50.215834Z","shell.execute_reply.started":"2023-08-26T09:32:43.749048Z","shell.execute_reply":"2023-08-26T09:32:50.214632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images = glob.glob('/kaggle/input/rsna-breast-cancer-512-pngs/*.png')\n\ntrain_images[0]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:32:54.517693Z","iopub.execute_input":"2023-08-26T09:32:54.518728Z","iopub.status.idle":"2023-08-26T09:32:56.046264Z","shell.execute_reply.started":"2023-08-26T09:32:54.518682Z","shell.execute_reply":"2023-08-26T09:32:56.045263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for f in tqdm(train_images[-6:]):\n    img = cv2.imread(f)\n    plt.figure(figsize=(5, 5))\n    plt.imshow(img, cmap=\"gray\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:32:59.650118Z","iopub.execute_input":"2023-08-26T09:32:59.651035Z","iopub.status.idle":"2023-08-26T09:33:01.508449Z","shell.execute_reply.started":"2023-08-26T09:32:59.650991Z","shell.execute_reply":"2023-08-26T09:33:01.503732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntest_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\ntrain_df.head()\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:33:14.17211Z","iopub.execute_input":"2023-08-26T09:33:14.172532Z","iopub.status.idle":"2023-08-26T09:33:14.388053Z","shell.execute_reply.started":"2023-08-26T09:33:14.1725Z","shell.execute_reply":"2023-08-26T09:33:14.387053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images = glob.glob('/kaggle/input/rsna-breast-cancer-512-pngs/*.png')\n\nfor path in tqdm(train_images):\n    name = path.split('/')[-1]\n    chunks = name.split('.')[0]\n    patient_id = chunks.split('_')[0]\n    image_id = chunks.split('_')[1]\n    \n    idx = (train_df['patient_id']==int(patient_id)) & (train_df['image_id']==int(image_id))\n    train_df.loc[idx, 'img_path'] =path","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:33:21.457082Z","iopub.execute_input":"2023-08-26T09:33:21.457819Z","iopub.status.idle":"2023-08-26T09:34:01.780715Z","shell.execute_reply.started":"2023-08-26T09:33:21.45778Z","shell.execute_reply":"2023-08-26T09:34:01.779643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[['patient_id','image_id','img_path']].head()\ntrain_df['img_path'][0]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:34:05.796197Z","iopub.execute_input":"2023-08-26T09:34:05.797286Z","iopub.status.idle":"2023-08-26T09:34:05.813209Z","shell.execute_reply.started":"2023-08-26T09:34:05.797221Z","shell.execute_reply":"2023-08-26T09:34:05.811888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images = glob.glob('/kaggle/working/test_output/*.png')\n\nfor path in tqdm(test_images):\n    name = path.split('/')[-1]\n    chunks = name.split('.')[0]\n    patient_id = chunks.split('_')[0]\n    image_id = chunks.split('_')[1]\n    \n    idx = (test_df['patient_id']==int(patient_id)) & (test_df['image_id']==int(image_id))\n    test_df.loc[idx, 'img_path'] =path\n\ntest_df[['patient_id','image_id','img_path']].head()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:34:15.061022Z","iopub.execute_input":"2023-08-26T09:34:15.061411Z","iopub.status.idle":"2023-08-26T09:34:15.106259Z","shell.execute_reply.started":"2023-08-26T09:34:15.061379Z","shell.execute_reply":"2023-08-26T09:34:15.105177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature Engineering","metadata":{}},{"cell_type":"code","source":"cols = ['image_id','age','machine_id','img_path']\nfor i in list(train_df.drop(cols,axis=1).columns):\n    print(i)\n    print(train_df[i].value_counts())\n    print('----------------\\n')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:34:21.355562Z","iopub.execute_input":"2023-08-26T09:34:21.356013Z","iopub.status.idle":"2023-08-26T09:34:21.402141Z","shell.execute_reply.started":"2023-08-26T09:34:21.355973Z","shell.execute_reply":"2023-08-26T09:34:21.401275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Make a data frame to reduce bias in dataset non-cancer cases**","metadata":{}},{"cell_type":"code","source":"\ndf = train_df.copy()\n# imgs of cancer\ncanc_count = df.loc[df['cancer']==1].shape[0]\ncanc_count\n\n# pick as many non canc cases as canc cases\ndf2 = df.loc[df['cancer']==0][:canc_count]\n\n# use rest of imgs for testing model\ndf_test_1 = df.loc[df['cancer']==0][canc_count:]\n\n# see how is the split\ndf2['cancer'].value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:34:32.237968Z","iopub.execute_input":"2023-08-26T09:34:32.238348Z","iopub.status.idle":"2023-08-26T09:34:32.264703Z","shell.execute_reply.started":"2023-08-26T09:34:32.238317Z","shell.execute_reply":"2023-08-26T09:34:32.263627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df3 = df.loc[df['cancer']==1]\ndf4 = pd.concat([df2,df3],axis=0)\n# look at the split\ndf4['cancer'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:34:37.258604Z","iopub.execute_input":"2023-08-26T09:34:37.259476Z","iopub.status.idle":"2023-08-26T09:34:37.273989Z","shell.execute_reply.started":"2023-08-26T09:34:37.25944Z","shell.execute_reply":"2023-08-26T09:34:37.272709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perc_90 = int(1158*0.9)\n\nnoncanc_df = df4.loc[df4['cancer']==0]\n# take 90% non canc cases\nnoncanc_df_train = noncanc_df[:perc_90]\n# take 10% non canc cases\nnoncanc_df_test = noncanc_df[perc_90:]\n\ncanc_df = df4.loc[df4['cancer']==1]\n# take 90% canc cases\ncanc_df_train = canc_df[:perc_90]\n# take 10% canc cases\ncanc_df_test = canc_df[perc_90:]\n\ntrain_df_new = pd.concat([noncanc_df_train,canc_df_train],axis = 0)\ntest_df_new = pd.concat([noncanc_df_test,canc_df_test],axis = 0)\ntrain_df_new['cancer'].value_counts()\ntest_df_new['cancer'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:34:42.016741Z","iopub.execute_input":"2023-08-26T09:34:42.017105Z","iopub.status.idle":"2023-08-26T09:34:42.039108Z","shell.execute_reply.started":"2023-08-26T09:34:42.017074Z","shell.execute_reply":"2023-08-26T09:34:42.038224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_train = list(train_df_new.drop(['img_path'], axis=1).columns)\nY_val = list(test_df_new.drop(['img_path'], axis=1).columns)\n# Y_test = list(test_df.drop(['img_path'], axis=1).columns)\n# unq_disease = len(Y_val)\n# unq_disease\nY_train","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:34:47.395687Z","iopub.execute_input":"2023-08-26T09:34:47.396086Z","iopub.status.idle":"2023-08-26T09:34:47.406321Z","shell.execute_reply.started":"2023-08-26T09:34:47.396057Z","shell.execute_reply":"2023-08-26T09:34:47.405104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255,\n                                                                horizontal_flip=True,\n                                                                vertical_flip=True,\n                                                                rotation_range=90,\n                                                               validation_split=0.20)\ntest_datagen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255)\n# The value for class_mode in flow_from_dataframe MUST be 'raw' if you are attempting to do multilabel classification.\ntrain_gen = train_datagen.flow_from_dataframe(train_df_new, \n                                              x_col='img_path', \n                                              y_col=['cancer'],\n                                              target_size=(128,128),\n                                              class_mode='raw',\n                                              batch_size=16,\n                                              shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:59:02.348917Z","iopub.execute_input":"2023-08-26T09:59:02.349334Z","iopub.status.idle":"2023-08-26T09:59:04.709434Z","shell.execute_reply.started":"2023-08-26T09:59:02.349299Z","shell.execute_reply":"2023-08-26T09:59:04.708507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_gen = train_datagen.flow_from_dataframe(test_df_new,\n                                          x_col='img_path',\n                                          y_col=['cancer'],\n                                          target_size=(128,128),\n                                          class_mode='raw',\n                                          batch_size=8)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:59:15.340491Z","iopub.execute_input":"2023-08-26T09:59:15.340872Z","iopub.status.idle":"2023-08-26T09:59:15.655521Z","shell.execute_reply.started":"2023-08-26T09:59:15.340839Z","shell.execute_reply":"2023-08-26T09:59:15.654412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen = train_datagen.flow_from_dataframe(test_df_new,\n                                            x_col='img_path',\n                                            y_col='cancer',\n                                          target_size=(128,128),\n                                          class_mode='raw',\n                                          batch_size=8,\n                                         seed = 42)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:59:31.857728Z","iopub.execute_input":"2023-08-26T09:59:31.858095Z","iopub.status.idle":"2023-08-26T09:59:31.871286Z","shell.execute_reply.started":"2023-08-26T09:59:31.858064Z","shell.execute_reply":"2023-08-26T09:59:31.870339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Image Augmentation**","metadata":{}},{"cell_type":"code","source":"# Load some sample images\nimg_path = '/kaggle/input/rsna-breast-cancer-512-pngs/10006_462822612.png'\nimg = tf.keras.preprocessing.image.load_img(img_path, target_size=(128, 128))\n\n# Convert image to numpy array\nimg_array = tf.keras.preprocessing.image.img_to_array(img)\n\n# Add a batch dimension to the array\nimg_array = np.expand_dims(img_array, axis=0)\n\n# Generate augmented images using the flow() method\naug_iter = train_datagen.flow(img_array, batch_size=1)\n\n# Visualize the augmented images\nfig, ax = plt.subplots(1, 5, figsize=(15, 15))\nfor i in range(5):\n    aug_img = next(aug_iter)[0]\n    ax[i].imshow(aug_img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:59:47.641784Z","iopub.execute_input":"2023-08-26T09:59:47.642212Z","iopub.status.idle":"2023-08-26T09:59:48.41285Z","shell.execute_reply.started":"2023-08-26T09:59:47.642178Z","shell.execute_reply":"2023-08-26T09:59:48.411756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Modelling**","metadata":{}},{"cell_type":"code","source":"def UNet(inputs):\n    # First convolution block\n    x = tf.keras.layers.Conv2D(64, 3, activation='relu', padding='same', kernel_initializer='he_normal')(inputs)\n    d1_con = tf.keras.layers.Conv2D(64, 3, activation='relu', padding='same', kernel_initializer='he_normal')(x)\n    d1 = tf.keras.layers.MaxPool2D(pool_size=2, strides=2)(d1_con)\n    \n    # Second convolution block\n    d2 = tf.keras.layers.Conv2D(128, 3, activation='relu', padding='same', kernel_initializer='he_normal')(d1)\n    d2_con = tf.keras.layers.Conv2D(128, 3, activation='relu', padding='same', kernel_initializer='he_normal')(d2)\n\n    d2 = tf.keras.layers.MaxPool2D(pool_size=2, strides=2)(d2_con)\n    \n    # Third convolution block\n    d3 = tf.keras.layers.Conv2D(256, 3, activation='relu', padding='same', kernel_initializer='he_normal')(d2)\n    d3_con = tf.keras.layers.Conv2D(256, 3, activation='relu', padding='same', kernel_initializer='he_normal')(d3)\n    d3 = tf.keras.layers.MaxPool2D(pool_size=2, strides=2)(d3_con)\n    \n     # Fourth convolution block\n    d4 = tf.keras.layers.Conv2D(512, 3, activation='relu', padding='same', kernel_initializer='he_normal')(d3)\n    d4_con = tf.keras.layers.Conv2D(512, 3, activation='relu', padding='same', kernel_initializer='he_normal')(d4)\n    d4 = tf.keras.layers.MaxPool2D(pool_size=2, strides=2)(d4_con)\n    \n    # Bottleneck layer\n    b = tf.keras.layers.Conv2D(1024, 3, activation='relu', padding='same', kernel_initializer='he_normal')(d4)\n    b = tf.keras.layers.Conv2D(1024, 3, activation='relu', padding='same', kernel_initializer='he_normal')(b)\n    \n    # First upsampling block\n    u1 = tf.keras.layers.Conv2DTranspose(512, 3, strides =(2,2),padding='same')(b)\n    u1 = tf.keras.layers.Concatenate(axis=3)([u1, d4_con])\n    u1 = tf.keras.layers.Conv2D(512, 3, activation='relu', padding='same', kernel_initializer='he_normal')(u1)\n    u1 = tf.keras.layers.Conv2D(512, 3, activation='relu', padding='same', kernel_initializer='he_normal')(u1)\n    \n    # Second upsampling block\n    u2 = tf.keras.layers.Conv2DTranspose(256, 3, strides =(2,2),padding='same')(u1)\n    u2 = tf.keras.layers.Concatenate(axis=3)([u2, d3_con])\n    u2 = tf.keras.layers.Conv2D(256, 3, activation='relu', padding='same', kernel_initializer='he_normal')(u2)\n    u2 = tf.keras.layers.Conv2D(256, 3, activation='relu', padding='same', kernel_initializer='he_normal')(u2)\n    \n     # Third upsampling block\n    u3 = tf.keras.layers.Conv2DTranspose(128, 3, strides =(2,2),padding='same')(u2)\n    u3 = tf.keras.layers.Concatenate(axis=3)([u3, d2_con])\n    u3 = tf.keras.layers.Conv2D(128, 3, activation='relu', padding='same', kernel_initializer='he_normal')(u3)\n    u3 = tf.keras.layers.Conv2D(128, 3, activation='relu', padding='same', kernel_initializer='he_normal')(u3)\n    \n    # Fourth upsampling block\n    u4 = tf.keras.layers.Conv2DTranspose(64, 3, strides =(2,2),padding='same')(u3)\n    u4 = tf.keras.layers.Concatenate(axis=3)([u4, d1_con])\n    u4 = tf.keras.layers.Conv2D(64, 3, activation='relu', padding='same', kernel_initializer='he_normal')(u4)\n    u4 = tf.keras.layers.Conv2D(64, 3, activation='relu', padding='same', kernel_initializer='he_normal')(u4)\n    \n     \n    # Flatten and output\n    flat = tf.keras.layers.Flatten()(u4)\n    hid1 = tf.keras.layers.Dense(units=50, activation='relu')(flat)\n    out = tf.keras.layers.Dense(units=1, activation='softmax')(hid1)\n    model = tf.keras.Model(inputs=[inputs], outputs=[out])\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-08-26T09:55:08.370935Z","iopub.execute_input":"2023-08-26T09:55:08.371352Z","iopub.status.idle":"2023-08-26T09:55:08.393534Z","shell.execute_reply.started":"2023-08-26T09:55:08.371316Z","shell.execute_reply":"2023-08-26T09:55:08.392283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.metrics import Accuracy\n\nauc = tf.keras.metrics.AUC(multi_label=True,thresholds=[0,0.5])\naucpr = tf.keras.metrics.AUC(curve='PR',multi_label=True,thresholds=[0,0.5])\ninputs = tf.keras.layers.Input(shape=(128,128,3))\nunet = UNet(inputs)\nunet.compile(optimizer='adam', loss='binary_crossentropy', metrics=[auc,aucpr]) #,tf.keras.metrics.Accuracy()\nunet.summary()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T10:00:15.383113Z","iopub.execute_input":"2023-08-26T10:00:15.383897Z","iopub.status.idle":"2023-08-26T10:00:15.764352Z","shell.execute_reply.started":"2023-08-26T10:00:15.38386Z","shell.execute_reply":"2023-08-26T10:00:15.763564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_history = unet.fit(train_gen, epochs=5, validation_data=val_gen)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T10:00:21.630096Z","iopub.execute_input":"2023-08-26T10:00:21.631113Z","iopub.status.idle":"2023-08-26T10:04:22.057122Z","shell.execute_reply.started":"2023-08-26T10:00:21.631075Z","shell.execute_reply":"2023-08-26T10:04:22.055897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model loss**","metadata":{}},{"cell_type":"code","source":"plt.plot(model_history.history['loss'], color='red')\nplt.plot(model_history.history['val_loss'], color='green')\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\n\n\n\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T10:06:15.939972Z","iopub.execute_input":"2023-08-26T10:06:15.940377Z","iopub.status.idle":"2023-08-26T10:06:16.271288Z","shell.execute_reply.started":"2023-08-26T10:06:15.940344Z","shell.execute_reply":"2023-08-26T10:06:16.270161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate model on test data\nevaluation_metrics = unet.evaluate(val_gen)\n\nevaluation_metrics\n\n# for train_gen : loss: 0.6638 - accuracy: 0.6247 - auc_10: 0.5000 - auc_11: 0.6247\n# for val_gen : loss: 1.1809 - accuracy: 0.0000e+00 - auc_10: 0.0000e+00 - auc_11: 0.0000e+00\n# for test_gen : loss: 0.7860 - accuracy: 0.5000 - auc_10: 0.5000 - auc_11: 0.5000","metadata":{"execution":{"iopub.status.busy":"2023-08-26T10:06:41.770042Z","iopub.execute_input":"2023-08-26T10:06:41.770453Z","iopub.status.idle":"2023-08-26T10:06:44.441253Z","shell.execute_reply.started":"2023-08-26T10:06:41.770419Z","shell.execute_reply":"2023-08-26T10:06:44.440283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_temp1 = df4.loc[df4['cancer']==0]\nnon_canc_path = df_temp1['img_path'].iloc[0]\nnon_canc_path","metadata":{"execution":{"iopub.status.busy":"2023-08-26T10:09:06.163107Z","iopub.execute_input":"2023-08-26T10:09:06.164087Z","iopub.status.idle":"2023-08-26T10:09:06.174803Z","shell.execute_reply.started":"2023-08-26T10:09:06.164052Z","shell.execute_reply":"2023-08-26T10:09:06.173619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing import image\nimport numpy as np\n\nimg_path = non_canc_path\nimg = image.load_img(img_path, target_size=(128, 128))\nimg_array = image.img_to_array(img)\nimg_array = np.expand_dims(img_array, axis=0)\nimg_array = img_array / 255.0 # normalize the pixel value\n\n# make a prediction on the input image\npreds = unet.predict(img_array)\n\n# get the predicted class by taking the argmax of the output\npredicted_class = np.argmax(preds, axis=-1)[0]\n\n# print the predicted class\nprint('Predicted class:', predicted_class)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-26T10:09:27.70687Z","iopub.execute_input":"2023-08-26T10:09:27.707252Z","iopub.status.idle":"2023-08-26T10:09:28.806951Z","shell.execute_reply.started":"2023-08-26T10:09:27.707205Z","shell.execute_reply":"2023-08-26T10:09:28.805869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}