{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Here in this notebook I use Transfer Learning for this challange.\n# Cpoy and edit as your wish and\n# Upvote it to encourage me. Thank You!","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport numpy as np\nimport tensorflow as tf\nimport os\nimport pandas as pd, numpy as np, random, shutil\nimport tensorflow as tf, re, math\nimport tensorflow.keras.backend as K\nimport sklearn\nimport matplotlib.pyplot as plt\nimport tensorflow_addons as tfa\nimport tensorflow_probability as tfp\nimport wandb\nimport yaml\n\nfrom IPython import display as ipd\nfrom glob import glob\nfrom tqdm import tqdm\nfrom sklearn.model_selection import KFold, StratifiedKFold, GroupKFold, StratifiedGroupKFold\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.utils.class_weight import compute_class_weight\n\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nfrom sklearn.utils import shuffle\nfrom sklearn.utils import class_weight\nfrom sklearn.preprocessing import minmax_scale\nimport random\nimport cv2\nfrom imgaug import augmenters as iaa\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Dense, Dropout, Activation, Input, BatchNormalization, GlobalAveragePooling2D\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nfrom tensorflow.keras.experimental import CosineDecay\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.applications import EfficientNetB3\nfrom tensorflow.keras.layers.experimental.preprocessing import RandomCrop,CenterCrop, RandomRotation","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport cv2\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tqdm import tqdm\nimport os\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.applications.vgg16 import VGG16\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, TensorBoard, ModelCheckpoint\nfrom sklearn.metrics import classification_report,confusion_matrix\nimport ipywidgets as widgets\nimport io\nfrom PIL import Image\nfrom IPython.display import display,clear_output\nfrom warnings import filterwarnings","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:08:51.400841Z","iopub.execute_input":"2023-01-05T17:08:51.401215Z","iopub.status.idle":"2023-01-05T17:08:51.409067Z","shell.execute_reply.started":"2023-01-05T17:08:51.401172Z","shell.execute_reply":"2023-01-05T17:08:51.407864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(10)\n\ntrain_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n\ntest_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\n\nbase_path='/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/'\n\n# saving image path into train dataframe\ntrain_df['img_path']= f'{base_path}/train_images_processed_cv2_256'\\\n                    + '/' + train_df.patient_id.astype(str)\\\n                    + '/' + train_df.image_id.astype(str)\\\n                    + '.png'\n\n\n\ndisplay(train_df.head(3))","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:06:32.42712Z","iopub.execute_input":"2023-01-05T17:06:32.427866Z","iopub.status.idle":"2023-01-05T17:06:32.642191Z","shell.execute_reply.started":"2023-01-05T17:06:32.427828Z","shell.execute_reply":"2023-01-05T17:06:32.641193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the counts for each class\ncases_count = train_df['cancer'].value_counts()\nprint(cases_count)\n\n# Plot the results \nplt.figure(figsize=(10,8))\nsns.barplot(x=cases_count.index, y= cases_count.values)\nplt.title('Number of cases', fontsize=14)\nplt.xlabel('Case type', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.xticks(range(len(cases_count.index)), ['Normal(0)', 'Cancer(1)'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:06:38.186054Z","iopub.execute_input":"2023-01-05T17:06:38.186504Z","iopub.status.idle":"2023-01-05T17:06:38.399187Z","shell.execute_reply.started":"2023-01-05T17:06:38.186471Z","shell.execute_reply":"2023-01-05T17:06:38.398237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Cancer_samples = (train_df[train_df['cancer']==1]['img_path'].iloc[0:5]).tolist()\nNormal_samples = (train_df[train_df['cancer']==0]['img_path'].iloc[0:5]).tolist()\n# Concat the data in a single list and del the above two list\nsamples = Cancer_samples + Normal_samples\n# source = \"../input/melanoma-merged-external-data-512x512-jpeg/512x512-dataset-melanoma/512x512-dataset-melanoma/\"\n# Plot the data \nf, ax = plt.subplots(2,5, figsize=(30,10))\nfor i in range(10):\n    img = tf.keras.preprocessing.image.load_img(samples[i])\n    ax[i//5, i%5].imshow(img, cmap='gray')\n    if i<5:\n        ax[i//5, i%5].set_title(\"Cancer\")\n    else:\n        ax[i//5, i%5].set_title(\"Normal\")\n    ax[i//5, i%5].axis('off')\n    ax[i//5, i%5].set_aspect('auto')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:32:07.367123Z","iopub.execute_input":"2023-01-05T17:32:07.367828Z","iopub.status.idle":"2023-01-05T17:32:08.080942Z","shell.execute_reply.started":"2023-01-05T17:32:07.367793Z","shell.execute_reply":"2023-01-05T17:32:08.08003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_augmentation_layers = tf.keras.Sequential(\n    [\n        layers.experimental.preprocessing.RandomCrop(height=256, width=256),\n        layers.experimental.preprocessing.RandomFlip(\"horizontal_and_vertical\"),\n        layers.experimental.preprocessing.RandomRotation(0.25),\n        layers.experimental.preprocessing.RandomZoom((-0.2, 0)),\n        layers.experimental.preprocessing.RandomContrast((0.2,0.2)),\n])\n","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:32:31.256934Z","iopub.execute_input":"2023-01-05T17:32:31.257333Z","iopub.status.idle":"2023-01-05T17:32:31.27617Z","shell.execute_reply.started":"2023-01-05T17:32:31.257298Z","shell.execute_reply":"2023-01-05T17:32:31.275333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = tf.keras.preprocessing.image.load_img(train_df['img_path'][90])\n\n\nimage = tf.expand_dims(np.array(image), 0)\n\nprint(image.shape)\n\nplt.figure(figsize=(10, 10))\nfor i in range(6):\n  augmented_image = data_augmentation_layers(image)\n  print('augmented_image ',augmented_image.shape)\n  ax = plt.subplot(3, 3, i + 1)\n  plt.imshow(augmented_image[0])\n  plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:32:34.108001Z","iopub.execute_input":"2023-01-05T17:32:34.110866Z","iopub.status.idle":"2023-01-05T17:32:34.977078Z","shell.execute_reply.started":"2023-01-05T17:32:34.110808Z","shell.execute_reply":"2023-01-05T17:32:34.976112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imgaug import augmenters as iaa\n\n\n\nclass DataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, df, path, batch_size=32, shuffle=True,aug=True,labels=True):\n        self.df = df.copy()\n        if 'prediction_id' not in df:\n            self.df['prediction_id'] = df[\"patient_id\"].astype(str) + '_' + df[\"laterality\"].astype(str)\n\n        self.prediction_ids = self.df['prediction_id'].unique()\n        self.labels = labels\n        if self.labels ==True:\n            self.labels = self.df.groupby('prediction_id')['cancer'].max()\n        self.path = path\n        self.batch_size = batch_size\n        self.aug=aug\n        self.shuffle = shuffle\n        self.on_epoch_end()\n\n    def __len__(self):\n        \"\"\"Denotes the number of batches per epoch\"\"\"\n        return int(len(self.prediction_ids) / self.batch_size)\n\n    def __getitem__(self, index):\n        \"\"\"Generate one batch of data\"\"\"\n        batch_indexes = self.prediction_ids[index * self.batch_size:(index + 1) * self.batch_size]\n        X, y = self.__data_generation(batch_indexes)\n        return X, y\n\n    def __get_input(self, path):\n        \n#         print('path   ',path)\n        image = tf.keras.preprocessing.image.load_img(path)\n        image_arr = tf.keras.preprocessing.image.img_to_array(image)\n        \n        if self.aug:\n            \n             image_arr=self.augmentor(image_arr)\n\n        \n        return image_arr\n\n    \n    def augmentor(self, images):\n        'Apply data augmentation'\n        images=data_augmentation_layers(images)\n\n        return images\n    \n    \n    def on_epoch_end(self):\n        \"\"\"Updates indexes after each epoch\"\"\"\n        if self.shuffle:\n            self.df = self.df.sample(frac=1).reset_index(drop=True)\n\n    def __data_generation(self, batch_indexes):\n        paths = self.get_paths_images(batch_indexes)\n        X = np.asarray([self.__get_input(path) for path in paths])\n        y = np.array([self.labels[batch_indexes]])\n        return X, y\n\n    def get_paths_images(self, batch_indexes):\n        batch = self.df[self.df['prediction_id'].isin(batch_indexes)]\n        rows_batch = self.get_rows(batch)\n        return self.path + rows_batch[\"patient_id\"].astype(str) + \"/\" + rows_batch[\"image_id\"].astype(\n            str) + \".png\"\n\n    def get_rows(self, batch):\n        \"\"\"Select only 1 MLO view picture per breast\"\"\"\n        only_MLO_view_images = batch[batch['view'] == 'MLO']\n        only_one_per_prediction_id = only_MLO_view_images.groupby('prediction_id')[['patient_id', 'image_id']].max()\n        return only_one_per_prediction_id","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:32:45.968593Z","iopub.execute_input":"2023-01-05T17:32:45.968953Z","iopub.status.idle":"2023-01-05T17:32:45.98498Z","shell.execute_reply.started":"2023-01-05T17:32:45.968923Z","shell.execute_reply":"2023-01-05T17:32:45.983973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nnp.random.seed(0)\n\nbatch_size=8\nepochs=2\nimg_path='/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/train_images_processed_cv2_256/'\n\ndef get_train_val_generator(train_size=0.8, batch_size=batch_size, filename=train_df, image_dir=img_path):\n\n    patient_ids = train_df[\"patient_id\"].unique()\n    np.random.shuffle(patient_ids)\n    train_size = int(len(patient_ids) * train_size)\n    train_ids = patient_ids[:train_size]\n    val_ids = patient_ids[train_size:]\n\n    df_train = train_df[train_df['patient_id'].isin(train_ids)]\n    df_val = train_df[train_df['patient_id'].isin(val_ids)]\n\n    train_gen = DataGenerator(df_train, batch_size=batch_size, path=image_dir,aug=True)\n    val_gen = DataGenerator(df_val, batch_size=batch_size, path=image_dir,aug=False)\n    \n    return train_gen, val_gen\n\ndataset_path=train_df\nimage_dir=img_path\ntrain_gen, val_gen = get_train_val_generator(filename=dataset_path, image_dir=image_dir)      \n\n     \n","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:32:52.921059Z","iopub.execute_input":"2023-01-05T17:32:52.921726Z","iopub.status.idle":"2023-01-05T17:32:53.024694Z","shell.execute_reply.started":"2023-01-05T17:32:52.921688Z","shell.execute_reply":"2023-01-05T17:32:53.023636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"3","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:33:00.47159Z","iopub.execute_input":"2023-01-05T17:33:00.471973Z","iopub.status.idle":"2023-01-05T17:33:00.501964Z","shell.execute_reply.started":"2023-01-05T17:33:00.471944Z","shell.execute_reply":"2023-01-05T17:33:00.500464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# vgg16\n","metadata":{}},{"cell_type":"code","source":"base_model = VGG16(input_shape = (256, 256, 3), # Shape of our images\ninclude_top = False, # Leave out the last fully connected layer\nweights = 'imagenet')","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:33:50.687549Z","iopub.execute_input":"2023-01-05T17:33:50.687996Z","iopub.status.idle":"2023-01-05T17:33:50.961195Z","shell.execute_reply.started":"2023-01-05T17:33:50.687955Z","shell.execute_reply":"2023-01-05T17:33:50.960243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model.trainable=False","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:33:58.052939Z","iopub.execute_input":"2023-01-05T17:33:58.053335Z","iopub.status.idle":"2023-01-05T17:33:58.058697Z","shell.execute_reply.started":"2023-01-05T17:33:58.053302Z","shell.execute_reply":"2023-01-05T17:33:58.057734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1 = base_model.output\n\nmodel1 = tf.keras.layers.Dense(512,activation='relu')(model1)\nmodel1 = tf.keras.layers.GlobalAveragePooling2D()(model1)\nmodel1 = tf.keras.layers.Dropout(rate=0.5)(model1)\nmodel1 = tf.keras.layers.Dense(1,activation='sigmoid')(model1)\nmodel1 = tf.keras.models.Model(inputs=base_model.input, outputs = model1)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:35:02.795423Z","iopub.execute_input":"2023-01-05T17:35:02.795781Z","iopub.status.idle":"2023-01-05T17:35:02.834371Z","shell.execute_reply.started":"2023-01-05T17:35:02.795752Z","shell.execute_reply":"2023-01-05T17:35:02.833481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:35:15.575386Z","iopub.execute_input":"2023-01-05T17:35:15.575782Z","iopub.status.idle":"2023-01-05T17:35:15.58377Z","shell.execute_reply.started":"2023-01-05T17:35:15.57575Z","shell.execute_reply":"2023-01-05T17:35:15.582729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tensorboard1 = TensorBoard(log_dir = 'logs')\ncheckpoint1 = ModelCheckpoint(\"vgg16\",monitor=\"val_accuracy\",save_best_only=True,mode=\"auto\",verbose=1)\nreduce_lr1 = ReduceLROnPlateau(monitor = 'val_accuracy', factor = 0.4, patience = 2, min_delta = 0.0001,\n                              mode='auto',verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:30:27.750657Z","iopub.execute_input":"2023-01-05T17:30:27.751017Z","iopub.status.idle":"2023-01-05T17:30:28.033686Z","shell.execute_reply.started":"2023-01-05T17:30:27.750986Z","shell.execute_reply":"2023-01-05T17:30:28.032592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1.compile(optimizer=\"Adam\",loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:35:32.734174Z","iopub.execute_input":"2023-01-05T17:35:32.73461Z","iopub.status.idle":"2023-01-05T17:35:32.747238Z","shell.execute_reply.started":"2023-01-05T17:35:32.734574Z","shell.execute_reply":"2023-01-05T17:35:32.74617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model1.fit(train_gen, validation_data=val_gen, epochs=5,callbacks=[tensorboard1,checkpoint1,reduce_lr1])\n","metadata":{"execution":{"iopub.status.busy":"2023-01-05T17:35:36.529843Z","iopub.execute_input":"2023-01-05T17:35:36.530229Z","iopub.status.idle":"2023-01-05T18:20:22.875269Z","shell.execute_reply.started":"2023-01-05T17:35:36.530181Z","shell.execute_reply":"2023-01-05T18:20:22.874245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Images\n\nDATASET_PATH='/kaggle/input/rsna-breast-cancer-detection/'\ntest_df = pd.read_csv(os.path.join(DATASET_PATH, \"test.csv\"))\ndisplay(test_df.head())\nprint(f'cases: {len(test_df)}')\n\n# Show sample submission example\n\ndf_sub = pd.read_csv(os.path.join(DATASET_PATH, \"sample_submission.csv\"))\ndisplay(df_sub.head())\nprint(f'cases: {len(df_sub)}')","metadata":{"execution":{"iopub.status.busy":"2023-01-05T18:20:36.11263Z","iopub.execute_input":"2023-01-05T18:20:36.113366Z","iopub.status.idle":"2023-01-05T18:20:36.165366Z","shell.execute_reply.started":"2023-01-05T18:20:36.113309Z","shell.execute_reply":"2023-01-05T18:20:36.164445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(image_path):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels = 3)\n    img = tf.image.resize(img, [256, 256])\n    img = tf.cast(img, dtype = tf.float32)\n    img = img/255.0\n    return img\n\n# test_paths=[]\n# test_dir='/kaggle/input/rsnatest/test_images_256/10008/'\n# img_path= os.listdir (test_dir)\n\n\n\nDF_PATH = '/kaggle/input/rsna-breast-cancer-detection'\ndf = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\n\ntest_path='/kaggle/input/rsnatest/test_images_256/'\n\ntest_dir = f'{test_path}'\n\n\n\n\ntest_df['img_path']= f'{test_dir}/'\\\n                    + '/' + test_df.patient_id.astype(str)\\\n                    + '/' + test_df.image_id.astype(str)\\\n                    + '.png'\n\n\ntest_df\n# image = tf.keras.preprocessing.image.load_img(test_df['img_path'][1])\n        \n# plt.imshow(image)\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-05T18:20:39.849019Z","iopub.execute_input":"2023-01-05T18:20:39.84939Z","iopub.status.idle":"2023-01-05T18:20:39.875089Z","shell.execute_reply.started":"2023-01-05T18:20:39.849358Z","shell.execute_reply":"2023-01-05T18:20:39.874222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds=[]\nfor i in range (len(test_df['img_path'])):\n    \n    image = tf.keras.preprocessing.image.load_img(test_df['img_path'][i])\n\n    image=tf.expand_dims(np.array(image), 0)\n    pred=model1.predict(np.asarray(image))\n    \n    preds.append(pred)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T18:20:59.141341Z","iopub.execute_input":"2023-01-05T18:20:59.141703Z","iopub.status.idle":"2023-01-05T18:20:59.737841Z","shell.execute_reply.started":"2023-01-05T18:20:59.141672Z","shell.execute_reply":"2023-01-05T18:20:59.736911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.DataFrame({'prediction_id':test_df.prediction_id,\n                        'cancer':preds})\n\npred_df['cancer']=(pred_df.cancer > 0.5).astype(int)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-01-05T18:21:04.620893Z","iopub.execute_input":"2023-01-05T18:21:04.621327Z","iopub.status.idle":"2023-01-05T18:21:04.628835Z","shell.execute_reply.started":"2023-01-05T18:21:04.621288Z","shell.execute_reply":"2023-01-05T18:21:04.627867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-05T18:52:50.402156Z","iopub.execute_input":"2023-01-05T18:52:50.402543Z","iopub.status.idle":"2023-01-05T18:52:50.409615Z","shell.execute_reply.started":"2023-01-05T18:52:50.402512Z","shell.execute_reply":"2023-01-05T18:52:50.408442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}