{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":4696088,"sourceType":"datasetVersion","datasetId":2687741},{"sourceId":4732257,"sourceType":"datasetVersion","datasetId":2738442}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:21:55.520731Z","iopub.execute_input":"2024-05-18T01:21:55.521173Z","iopub.status.idle":"2024-05-18T01:21:55.532696Z","shell.execute_reply.started":"2024-05-18T01:21:55.521141Z","shell.execute_reply":"2024-05-18T01:21:55.531689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow==2.15.0","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:21:55.534322Z","iopub.execute_input":"2024-05-18T01:21:55.534587Z","iopub.status.idle":"2024-05-18T01:22:07.835014Z","shell.execute_reply.started":"2024-05-18T01:21:55.534564Z","shell.execute_reply":"2024-05-18T01:22:07.834102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nprint(tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:07.836293Z","iopub.execute_input":"2024-05-18T01:22:07.836564Z","iopub.status.idle":"2024-05-18T01:22:11.452457Z","shell.execute_reply.started":"2024-05-18T01:22:07.836539Z","shell.execute_reply":"2024-05-18T01:22:11.451481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow-addons==0.21.0","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:11.455068Z","iopub.execute_input":"2024-05-18T01:22:11.455909Z","iopub.status.idle":"2024-05-18T01:22:23.407685Z","shell.execute_reply.started":"2024-05-18T01:22:11.455849Z","shell.execute_reply":"2024-05-18T01:22:23.406694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport numpy as np\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nimport os\nimport pandas as pd, numpy as np, random, shutil\nimport tensorflow as tf, re, math\nimport tensorflow.keras.backend as K\nimport sklearn\nimport matplotlib.pyplot as plt\nimport tensorflow_probability as tfp\nimport wandb\nimport yaml\n\nfrom IPython import display as ipd\nfrom glob import glob\nfrom tqdm import tqdm\nfrom sklearn.model_selection import KFold, StratifiedKFold, GroupKFold, StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.utils.class_weight import compute_class_weight\n\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nfrom sklearn.utils import shuffle\nfrom sklearn.utils import class_weight\nfrom sklearn.preprocessing import minmax_scale\nimport random\nimport cv2\nfrom imgaug import augmenters as iaa\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Dense, Dropout, Activation, Input, BatchNormalization, GlobalAveragePooling2D\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nfrom tensorflow.keras.experimental import CosineDecay\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.applications import EfficientNetB3\nfrom tensorflow.keras.layers.experimental.preprocessing import RandomCrop,CenterCrop, RandomRotation","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:23.409028Z","iopub.execute_input":"2024-05-18T01:22:23.409359Z","iopub.status.idle":"2024-05-18T01:22:24.914428Z","shell.execute_reply.started":"2024-05-18T01:22:23.409331Z","shell.execute_reply":"2024-05-18T01:22:24.913603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport cv2\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tqdm import tqdm\nimport os\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.applications.vgg16 import VGG16\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, TensorBoard, ModelCheckpoint\nfrom sklearn.metrics import classification_report,confusion_matrix\nimport ipywidgets as widgets\nimport io\nfrom PIL import Image\nfrom IPython.display import display,clear_output\nfrom warnings import filterwarnings","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:24.91563Z","iopub.execute_input":"2024-05-18T01:22:24.916635Z","iopub.status.idle":"2024-05-18T01:22:24.923405Z","shell.execute_reply.started":"2024-05-18T01:22:24.916607Z","shell.execute_reply":"2024-05-18T01:22:24.92252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(10)\n\ntrain_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n\ntest_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\n\nbase_path='/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/'\n\n# saving image path into train dataframe\ntrain_df['img_path']= f'{base_path}/train_images_processed_cv2_256'\\\n                    + '/' + train_df.patient_id.astype(str)\\\n                    + '/' + train_df.image_id.astype(str)\\\n                    + '.png'\n\n\n\ndisplay(train_df.head(3))\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:24.924537Z","iopub.execute_input":"2024-05-18T01:22:24.924835Z","iopub.status.idle":"2024-05-18T01:22:25.126527Z","shell.execute_reply.started":"2024-05-18T01:22:24.924808Z","shell.execute_reply":"2024-05-18T01:22:25.125502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the counts for each class\ncases_count = train_df['cancer'].value_counts()\nprint(cases_count)\n\n# Plot the results \nplt.figure(figsize=(5,4))\nsns.barplot(x=cases_count.index, y= cases_count.values)\nplt.title('Number of cases', fontsize=14)\nplt.xlabel('Case type', fontsize=4)\nplt.ylabel('Count', fontsize=4)\nplt.xticks(range(len(cases_count.index)), ['Normal(0)', 'Cancer(1)'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:25.127912Z","iopub.execute_input":"2024-05-18T01:22:25.128535Z","iopub.status.idle":"2024-05-18T01:22:25.362678Z","shell.execute_reply.started":"2024-05-18T01:22:25.128498Z","shell.execute_reply":"2024-05-18T01:22:25.361762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Cancer_samples = (train_df[train_df['cancer']==1]['img_path'].iloc[0:5]).tolist()\nNormal_samples = (train_df[train_df['cancer']==0]['img_path'].iloc[0:5]).tolist()\n# Concat the data in a single list and del the above two list\nsamples = Cancer_samples + Normal_samples\n# source = \"../input/melanoma-merged-external-data-512x512-jpeg/512x512-dataset-melanoma/512x512-dataset-melanoma/\"\n# Plot the data \nf, ax = plt.subplots(2,5, figsize=(30,10))\nfor i in range(10):\n    img = tf.keras.preprocessing.image.load_img(samples[i])\n    ax[i//5, i%5].imshow(img, cmap='gray')\n    if i<5:\n        ax[i//5, i%5].set_title(\"Cancer\")\n    else:\n        ax[i//5, i%5].set_title(\"Normal\")\n    ax[i//5, i%5].axis('off')\n    ax[i//5, i%5].set_aspect('auto')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:25.363943Z","iopub.execute_input":"2024-05-18T01:22:25.364246Z","iopub.status.idle":"2024-05-18T01:22:26.567829Z","shell.execute_reply.started":"2024-05-18T01:22:25.364221Z","shell.execute_reply":"2024-05-18T01:22:26.566913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_augmentation_layers = tf.keras.Sequential(\n    [\n        layers.experimental.preprocessing.RandomCrop(height=256, width=256),\n        layers.experimental.preprocessing.RandomFlip(\"horizontal_and_vertical\"),\n        layers.experimental.preprocessing.RandomRotation(0.25),\n        layers.experimental.preprocessing.RandomZoom((-0.2, 0)),\n        layers.experimental.preprocessing.RandomContrast((0.2,0.2)),\n])","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:26.571297Z","iopub.execute_input":"2024-05-18T01:22:26.5716Z","iopub.status.idle":"2024-05-18T01:22:27.148772Z","shell.execute_reply.started":"2024-05-18T01:22:26.571573Z","shell.execute_reply":"2024-05-18T01:22:27.147967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = tf.keras.preprocessing.image.load_img(train_df['img_path'][90])\n\n\nimage = tf.expand_dims(np.array(image), 0)\n\nprint(image.shape)\n\nplt.figure(figsize=(10, 10))\nfor i in range(6):\n  augmented_image = data_augmentation_layers(image)\n  print('augmented_image ',augmented_image.shape)\n  ax = plt.subplot(3, 3, i + 1)\n  plt.imshow(augmented_image[0])\n  plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:27.149787Z","iopub.execute_input":"2024-05-18T01:22:27.150083Z","iopub.status.idle":"2024-05-18T01:22:29.586063Z","shell.execute_reply.started":"2024-05-18T01:22:27.150057Z","shell.execute_reply":"2024-05-18T01:22:29.585077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imgaug import augmenters as iaa\n\n\n\nclass DataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, df, path, batch_size=32, shuffle=True,aug=True,labels=True):\n        self.df = df.copy()\n        if 'prediction_id' not in df:\n            self.df['prediction_id'] = df[\"patient_id\"].astype(str) + '_' + df[\"laterality\"].astype(str)\n\n        self.prediction_ids = self.df['prediction_id'].unique()\n        self.labels = labels\n        if self.labels ==True:\n            self.labels = self.df.groupby('prediction_id')['cancer'].max()\n        self.path = path\n        self.batch_size = batch_size\n        self.aug=aug\n        self.shuffle = shuffle\n        self.on_epoch_end()\n\n    def __len__(self):\n        \"\"\"Denotes the number of batches per epoch\"\"\"\n        return int(len(self.prediction_ids) / self.batch_size)\n\n    def __getitem__(self, index):\n        \"\"\"Generate one batch of data\"\"\"\n        batch_indexes = self.prediction_ids[index * self.batch_size:(index + 1) * self.batch_size]\n        X, y = self.__data_generation(batch_indexes)\n        return X, y\n\n    def __get_input(self, path):\n        \n#         print('path   ',path)\n        image = tf.keras.preprocessing.image.load_img(path)\n        image_arr = tf.keras.preprocessing.image.img_to_array(image)\n        \n        if self.aug:\n            \n             image_arr=self.augmentor(image_arr)\n\n        \n        return image_arr\n\n    \n    def augmentor(self, images):\n        'Apply data augmentation'\n        images=data_augmentation_layers(images)\n\n        return images\n    \n    \n    def on_epoch_end(self):\n        \"\"\"Updates indexes after each epoch\"\"\"\n        if self.shuffle:\n            self.df = self.df.sample(frac=1).reset_index(drop=True)\n\n    def __data_generation(self, batch_indexes):\n        paths = self.get_paths_images(batch_indexes)\n        X = np.asarray([self.__get_input(path) for path in paths])\n        y = np.array([self.labels[batch_indexes]])\n        return X, y\n\n    def get_paths_images(self, batch_indexes):\n        batch = self.df[self.df['prediction_id'].isin(batch_indexes)]\n        rows_batch = self.get_rows(batch)\n        return self.path + rows_batch[\"patient_id\"].astype(str) + \"/\" + rows_batch[\"image_id\"].astype(\n            str) + \".png\"\n\n    def get_rows(self, batch):\n        \"\"\"Select only 1 MLO view picture per breast\"\"\"\n        only_MLO_view_images = batch[batch['view'] == 'MLO']\n        only_one_per_prediction_id = only_MLO_view_images.groupby('prediction_id')[['patient_id', 'image_id']].max()\n        return only_one_per_prediction_id","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:29.587569Z","iopub.execute_input":"2024-05-18T01:22:29.588285Z","iopub.status.idle":"2024-05-18T01:22:29.605747Z","shell.execute_reply.started":"2024-05-18T01:22:29.588244Z","shell.execute_reply":"2024-05-18T01:22:29.60474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nnp.random.seed(0)\n\nbatch_size=8\nepochs=2\nimg_path='/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/train_images_processed_cv2_256/'\n\ndef get_train_val_generator(train_size=0.8, batch_size=batch_size, filename=train_df, image_dir=img_path):\n\n    patient_ids = train_df[\"patient_id\"].unique()\n    np.random.shuffle(patient_ids)\n    train_size = int(len(patient_ids) * train_size)\n    train_ids = patient_ids[:train_size]\n    val_ids = patient_ids[train_size:]\n\n    df_train = train_df[train_df['patient_id'].isin(train_ids)]\n    df_val = train_df[train_df['patient_id'].isin(val_ids)]\n\n    train_gen = DataGenerator(df_train, batch_size=batch_size, path=image_dir,aug=True)\n    val_gen = DataGenerator(df_val, batch_size=batch_size, path=image_dir,aug=False)\n    \n    return train_gen, val_gen\n\ndataset_path=train_df\nimage_dir=img_path\ntrain_gen, val_gen = get_train_val_generator(filename=dataset_path, image_dir=image_dir)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:29.607057Z","iopub.execute_input":"2024-05-18T01:22:29.607485Z","iopub.status.idle":"2024-05-18T01:22:29.738345Z","shell.execute_reply.started":"2024-05-18T01:22:29.607453Z","shell.execute_reply":"2024-05-18T01:22:29.737528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = VGG16(input_shape = (256, 256, 3), # Shape of our images\ninclude_top = False, # Leave out the last fully connected layer\nweights = 'imagenet')","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:29.739456Z","iopub.execute_input":"2024-05-18T01:22:29.739767Z","iopub.status.idle":"2024-05-18T01:22:30.147642Z","shell.execute_reply.started":"2024-05-18T01:22:29.73974Z","shell.execute_reply":"2024-05-18T01:22:30.146751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model.trainable=False","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:30.148786Z","iopub.execute_input":"2024-05-18T01:22:30.149095Z","iopub.status.idle":"2024-05-18T01:22:30.154572Z","shell.execute_reply.started":"2024-05-18T01:22:30.149069Z","shell.execute_reply":"2024-05-18T01:22:30.153517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1 = base_model.output\n\nmodel1 = tf.keras.layers.Dense(512,activation='relu')(model1)\nmodel1 = tf.keras.layers.GlobalAveragePooling2D()(model1)\nmodel1 = tf.keras.layers.Dropout(rate=0.5)(model1)\nmodel1 = tf.keras.layers.Dense(1,activation='sigmoid')(model1)\nmodel1 = tf.keras.models.Model(inputs=base_model.input, outputs = model1)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:30.155763Z","iopub.execute_input":"2024-05-18T01:22:30.15612Z","iopub.status.idle":"2024-05-18T01:22:30.213524Z","shell.execute_reply.started":"2024-05-18T01:22:30.156089Z","shell.execute_reply":"2024-05-18T01:22:30.212771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1.summary()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:30.214636Z","iopub.execute_input":"2024-05-18T01:22:30.214948Z","iopub.status.idle":"2024-05-18T01:22:30.268662Z","shell.execute_reply.started":"2024-05-18T01:22:30.214922Z","shell.execute_reply":"2024-05-18T01:22:30.267885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tensorboard1 = TensorBoard(log_dir = 'logs')\ncheckpoint1 = ModelCheckpoint(\"vgg16\",monitor=\"val_accuracy\",save_best_only=True,mode=\"auto\",verbose=1)\nreduce_lr1 = ReduceLROnPlateau(monitor = 'val_accuracy', factor = 0.4, patience = 2, min_delta = 0.0001,\n                              mode='auto',verbose=1)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:30.269795Z","iopub.execute_input":"2024-05-18T01:22:30.270068Z","iopub.status.idle":"2024-05-18T01:22:30.276599Z","shell.execute_reply.started":"2024-05-18T01:22:30.270045Z","shell.execute_reply":"2024-05-18T01:22:30.275665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1.compile(optimizer=\"Adam\",loss='binary_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:30.277951Z","iopub.execute_input":"2024-05-18T01:22:30.278787Z","iopub.status.idle":"2024-05-18T01:22:30.301024Z","shell.execute_reply.started":"2024-05-18T01:22:30.278758Z","shell.execute_reply":"2024-05-18T01:22:30.300174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model1.fit(train_gen, validation_data=val_gen, epochs=5,callbacks=[tensorboard1,checkpoint1,reduce_lr1])","metadata":{"execution":{"iopub.status.busy":"2024-05-18T01:22:30.301975Z","iopub.execute_input":"2024-05-18T01:22:30.302221Z","iopub.status.idle":"2024-05-18T02:44:07.595216Z","shell.execute_reply.started":"2024-05-18T01:22:30.302199Z","shell.execute_reply":"2024-05-18T02:44:07.594377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Images\n\nDATASET_PATH='/kaggle/input/rsna-breast-cancer-detection/'\ntest_df = pd.read_csv(os.path.join(DATASET_PATH, \"test.csv\"))\n# display(test_df.head())\n# print(f'cases: {len(test_df)}')","metadata":{"execution":{"iopub.status.busy":"2024-05-18T02:44:07.59644Z","iopub.execute_input":"2024-05-18T02:44:07.596697Z","iopub.status.idle":"2024-05-18T02:44:07.60462Z","shell.execute_reply.started":"2024-05-18T02:44:07.596675Z","shell.execute_reply":"2024-05-18T02:44:07.60364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(image_path):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels = 3)\n    img = tf.image.resize(img, [256, 256])\n    img = tf.cast(img, dtype = tf.float32)\n    img = img/255.0\n    return img\n\n# test_paths=[]\n# test_dir='/kaggle/input/rsnatest/test_images_256/10008/'\n# img_path= os.listdir (test_dir)\n\n\n\nDF_PATH = '/kaggle/input/rsna-breast-cancer-detection'\ndf = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\n\ntest_path='/kaggle/input/rsnatest/test_images_256/'\n\ntest_dir = f'{test_path}'\n\n\n\n\ntest_df['img_path']= f'{test_dir}/'\\\n                    + '/' + test_df.patient_id.astype(str)\\\n                    + '/' + test_df.image_id.astype(str)\\\n                    + '.png'\n\n\ntest_df\n# image = tf.keras.preprocessing.image.load_img(test_df['img_path'][1])\n        \n# plt.imshow(image)\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-18T02:44:07.605777Z","iopub.execute_input":"2024-05-18T02:44:07.606117Z","iopub.status.idle":"2024-05-18T02:44:07.62633Z","shell.execute_reply.started":"2024-05-18T02:44:07.606084Z","shell.execute_reply":"2024-05-18T02:44:07.625502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds=[]\nfor i in range (len(test_df['img_path'])):\n    \n    image = tf.keras.preprocessing.image.load_img(test_df['img_path'][i])\n\n    image=tf.expand_dims(np.array(image), 0)\n    pred=model1.predict(np.asarray(image))\n    \n    preds.append(pred)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T03:42:03.14934Z","iopub.execute_input":"2024-05-18T03:42:03.150451Z","iopub.status.idle":"2024-05-18T03:42:03.495417Z","shell.execute_reply.started":"2024-05-18T03:42:03.150409Z","shell.execute_reply":"2024-05-18T03:42:03.494438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.DataFrame({'prediction_id':test_df.prediction_id,\n                        'cancer':preds})\n\npred_df['cancer']=(pred_df.cancer > 0.5).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T03:42:41.611055Z","iopub.execute_input":"2024-05-18T03:42:41.611919Z","iopub.status.idle":"2024-05-18T03:42:41.617782Z","shell.execute_reply.started":"2024-05-18T03:42:41.611884Z","shell.execute_reply":"2024-05-18T03:42:41.61678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_df.columns)\nprint(pred_df.columns)","metadata":{"execution":{"iopub.status.busy":"2024-05-18T03:42:44.743673Z","iopub.execute_input":"2024-05-18T03:42:44.744535Z","iopub.status.idle":"2024-05-18T03:42:44.749608Z","shell.execute_reply.started":"2024-05-18T03:42:44.744499Z","shell.execute_reply":"2024-05-18T03:42:44.74871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to calculate probabilistic F1 score\ndef pfbeta(labels, predictions, beta):\n    y_true_count = 0  # Count of true positive labels\n    ctp = 0  # Cumulative true positives based on predictions\n    cfp = 0  # Cumulative false positives based on predictions\n\n    for idx in range(len(labels)):\n        # Clip predictions to be between 0 and 1\n        prediction = min(max(predictions[idx], 0), 1)\n        \n        if labels[idx]:\n            y_true_count += 1\n            ctp += prediction  # Add to cumulative true positives\n        else:\n            cfp += prediction  # Add to cumulative false positives\n\n    beta_squared = beta * beta\n    # Calculate precision and recall\n    if y_true_count > 0:  # Check if y_true_count is greater than zero\n        c_precision = ctp / (ctp + cfp)\n        c_recall = ctp / y_true_count\n\n        if c_precision > 0 and c_recall > 0:\n            # Calculate the F1 score with the given beta\n            result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n            return result\n    return 0 \npreds = []\nfor i in range(len(test_df['img_path'])):\n    img = tf.keras.preprocessing.image.load_img(test_df['img_path'][i], target_size=(256, 256))  # Adjust target_size if needed\n    img_array = np.array(img)\n    img_array = np.expand_dims(img_array, axis=0)\n    \n    pred = model1.predict(img_array)\n    preds.append(pred[0][0]) \n\npredictions = np.array(preds)\n\nlabels = pred_df['cancer'].values \n\nbeta = 1.0\n\n# Calculate the probabilistic F1 score\npf1_score = pfbeta(labels, predictions, beta)\n\nprint(f\"Probabilistic F1 Score: {pf1_score}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-05-18T03:42:50.035077Z","iopub.execute_input":"2024-05-18T03:42:50.035757Z","iopub.status.idle":"2024-05-18T03:42:50.345626Z","shell.execute_reply.started":"2024-05-18T03:42:50.035722Z","shell.execute_reply":"2024-05-18T03:42:50.344742Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_imgs = list(test_df.patient_id.astype(str) + '_' + test_df.laterality.astype(str))\n\nprint('Probabilistic F1 score:'+ str(pfbeta(labels_imgs, y_pred, 1)))","metadata":{"execution":{"iopub.status.busy":"2024-05-18T03:42:54.265425Z","iopub.execute_input":"2024-05-18T03:42:54.265919Z","iopub.status.idle":"2024-05-18T03:42:54.272823Z","shell.execute_reply.started":"2024-05-18T03:42:54.265875Z","shell.execute_reply":"2024-05-18T03:42:54.271839Z"},"trusted":true},"execution_count":null,"outputs":[]}]}