{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport time \nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nimport glob as gb\nimport cv2\nimport matplotlib.pyplot as plt\nimport pydicom as dcm\n\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed\n\nDATA_PATH = '/kaggle/input/rsna-breast-cancer-detection/'\nIMAGE_PATH = '/kaggle/input/rsna-breast-cancer-combined-images/output/'\nTEST_IMAGES = \"/kaggle/input/rsna-breast-cancer-detection/test_images/\"\n\ntry:\n    import pylibjpeg\nexcept:\n    !pip install /kaggle/input/rsna-2022-whl/{pydicom-2.3.0-py3-none-any.whl,pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-25T16:34:25.256149Z","iopub.execute_input":"2023-01-25T16:34:25.256469Z","iopub.status.idle":"2023-01-25T16:35:03.14154Z","shell.execute_reply.started":"2023-01-25T16:34:25.256384Z","shell.execute_reply":"2023-01-25T16:35:03.14032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install tensorflow==2.4.1\n# !apt install --allow-change-held-packages libcudnn8=8.1.0.77-1+cuda11.2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gpu_devices = tf.config.experimental.list_physical_devices(\"GPU\") \nfor device in gpu_devices: tf.config.experimental.set_memory_growth(device, True)","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:35:19.563437Z","iopub.execute_input":"2023-01-25T16:35:19.564442Z","iopub.status.idle":"2023-01-25T16:35:19.752195Z","shell.execute_reply.started":"2023-01-25T16:35:19.564403Z","shell.execute_reply":"2023-01-25T16:35:19.750818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tf_pfbeta(y_true, y_pred, beta = 1):\n    y_true = tf.cast(y_true, tf.float32)\n    y_true_count = tf.reduce_sum(y_true)\n    ctp = tf.reduce_sum(y_true * y_pred)\n    cfp = tf.reduce_sum((1 - y_true) * y_pred)\n    beta_squared = beta * beta\n    print(type(ctp), type(cfp), type(0))\n    c_precision = tf.where(ctp + cfp == 0, 0, ctp / (ctp + cfp))\n    c_recall =  tf.where(y_true_count == 0, 0, ctp / y_true_count)\n    return tf.where(c_precision + c_recall == 0, 0, (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall))\n\ndef check_breast_location(img):\n    left_half = img[:, :img.shape[1]//2]\n    right_half = img[:, img.shape[1]//2:]\n    if np.mean(left_half) > np.mean(right_half):\n        return \"L\"\n    else:\n        return \"R\"\n    \ndef read_and_process_dcm(image_path):\n    dicom = dcm.dcmread(image_path)\n    img = dicom.pixel_array\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n        \n    return img\n\ndef combine_patient_image(patient_id, df):\n        image_dict = {}\n        dcm_paths = df[df['patient_id'] == patient_id]['dcm_path'].values\n        for image_path in dcm_paths:\n            image_id = str(image_path).split('/')[-1].split('.')[0]\n            key = df[df['image_id'] == int(image_id)]['laterality_view'].values[0]\n            img = read_and_process_dcm(image_path)\n            img = cv2.resize(img, (256, 256))\n            breast_loc = check_breast_location(img)\n            if key[0] != breast_loc:\n                img = cv2.flip(img, 1)\n            image_dict[key] = img\n        mlo = np.hstack((image_dict['R_MLO'], image_dict['L_MLO']))\n        cc = np.hstack((image_dict['R_CC'], image_dict['L_CC']))\n        t = np.vstack((mlo, cc))\n        cv2.imwrite(\n                f'/kaggle/working/output/{patient_id}.png',\n                (t * 255).astype(np.uint8)\n        )\n","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:35:22.925347Z","iopub.execute_input":"2023-01-25T16:35:22.925702Z","iopub.status.idle":"2023-01-25T16:35:22.939324Z","shell.execute_reply.started":"2023-01-25T16:35:22.925671Z","shell.execute_reply":"2023-01-25T16:35:22.938345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(os.path.join(DATA_PATH, \"train.csv\"))\ntest = pd.read_csv(os.path.join(DATA_PATH, \"test.csv\"))\nsub = pd.read_csv(os.path.join(DATA_PATH, \"sample_submission.csv\"))\n\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:35:25.591598Z","iopub.execute_input":"2023-01-25T16:35:25.59197Z","iopub.status.idle":"2023-01-25T16:35:25.71978Z","shell.execute_reply.started":"2023-01-25T16:35:25.591941Z","shell.execute_reply":"2023-01-25T16:35:25.718816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['filename'] = train[\"patient_id\"].astype(str) + '_' + train[\"image_id\"].astype(str) + '.png'\ntrain['cancer'] = train['cancer'].astype(str)\n\ntest['laterality_view'] = test['laterality'] + \"_\"+ test['view']\ntest['dcm_path'] = TEST_IMAGES + test['patient_id'].astype(str) + '/' + test['image_id'].astype(str) + '.dcm'","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:35:27.235377Z","iopub.execute_input":"2023-01-25T16:35:27.235726Z","iopub.status.idle":"2023-01-25T16:35:27.34728Z","shell.execute_reply.started":"2023-01-25T16:35:27.235697Z","shell.execute_reply":"2023-01-25T16:35:27.346335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_df = train.groupby([\"patient_id\", \"laterality\"])['cancer'].max().unstack().reset_index()\ntest_only_first = test.groupby(['patient_id', 'view', 'laterality']).first().reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:35:28.759129Z","iopub.execute_input":"2023-01-25T16:35:28.759489Z","iopub.status.idle":"2023-01-25T16:35:30.980876Z","shell.execute_reply.started":"2023-01-25T16:35:28.759452Z","shell.execute_reply":"2023-01-25T16:35:30.97976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_df['image_path'] = IMAGE_PATH + train_labels_df['patient_id'].astype(str) + '.png'\ntest_only_first['image_path'] = '/kaggle/working/output/' + train_labels_df['patient_id'].astype(str) + '.png'\n\ntrain_labels = list(zip(train_labels_df['L'], train_labels_df['R']))\npath_and_labels = (train_labels_df['image_path'].values, train_labels)","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:35:30.98644Z","iopub.execute_input":"2023-01-25T16:35:30.988845Z","iopub.status.idle":"2023-01-25T16:35:31.04338Z","shell.execute_reply.started":"2023-01-25T16:35:30.988808Z","shell.execute_reply":"2023-01-25T16:35:31.042298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patients_ids = test.patient_id.unique()\n\nif not os.path.exists('/kaggle/working/output'):\n    !mkdir 'output'\n\n_ = Parallel(n_jobs=-1)(\n    delayed(combine_patient_image)(patient_id, test_only_first)\n    for patient_id in tqdm(patients_ids))","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:35:38.517205Z","iopub.execute_input":"2023-01-25T16:35:38.517564Z","iopub.status.idle":"2023-01-25T16:35:44.660849Z","shell.execute_reply.started":"2023-01-25T16:35:38.517532Z","shell.execute_reply":"2023-01-25T16:35:44.659419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_df['cancer'] = train_labels_df['L'].astype(str)+'_'+train_labels_df['R'].astype(str)\ntrain_labels_df['cancer'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:36:06.329302Z","iopub.execute_input":"2023-01-25T16:36:06.329684Z","iopub.status.idle":"2023-01-25T16:36:06.352873Z","shell.execute_reply.started":"2023-01-25T16:36:06.329648Z","shell.execute_reply":"2023-01-25T16:36:06.351794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def process_path(path, label):\n# #     print(path.to_string())\n#     img = tf.io.read_file(path)\n#     img = tf.io.decode_png(img, channels=3)\n#     img = tf.image.convert_image_dtype(img, tf.float32)\n#     img = tf.image.resize(img, (256, 256))\n    \n#     labels = {'L': label[0], 'R': label[1]}\n    \n#     return img, labels\n\n\n# AUTO = tf.data.AUTOTUNE\n# BATCH_SIZE = 32\n\n# data_augmentation = tf.keras.Sequential([\n#     tf.keras.layers.experimental.preprocessing.RandomZoom(\n#         height_factor=0.2, width_factor=0.2\n#     ),\n#     tf.keras.layers.experimental.preprocessing.RandomRotation(0.02)\n# ], name=\"data_augmentation\")\n\n# train_ds = tf.data.Dataset.from_tensor_slices(path_and_labels)\n# # train_ds = train_ds.shuffle(buffer_size=len(train_ds))\n# # train_ds = train_ds.map(process_path, num_parallel_calls=AUTO)\n\n# # train_ds.batch(BATCH_SIZE)\n\n# pipeline_train = (train_ds\n#                   .shuffle(BATCH_SIZE * 10)\n#                   .map(process_path, num_parallel_calls=AUTO)\n#                   .batch(BATCH_SIZE)\n#                   .map(lambda x, y: (data_augmentation(x), y), num_parallel_calls=AUTO)\n#                   .prefetch(AUTO)\n#                  )","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:36:09.431589Z","iopub.execute_input":"2023-01-25T16:36:09.431984Z","iopub.status.idle":"2023-01-25T16:36:09.437685Z","shell.execute_reply.started":"2023-01-25T16:36:09.431944Z","shell.execute_reply":"2023-01-25T16:36:09.436743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MyDataSet(tf.keras.utils.Sequence):\n    def __init__(self, df, image_path, shuffle=True, batch_size=32, labels=True):\n        self.df = df\n        self.image_path = image_path\n        self.patient_ids = self.df.patient_id.unique()\n        self.labels = labels\n        if self.labels: \n            self.target = self.df[['L', 'R']]\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.on_epoch_end()\n        \n    def __len__(self):\n        \"\"\"Denotes the number of batches per epoch\"\"\"\n        return int(len(self.patient_ids) / self.batch_size)\n    \n    def __getitem__(self, index):\n        batch_indexes = self.patient_ids[index * self.batch_size:(index + 1) * self.batch_size]\n        data_slice = self.df[self.df['patient_id'].isin(batch_indexes)]\n        paths = (self.image_path + data_slice.patient_id.astype(str) + '.png').unique()\n#         print(paths)\n        X = np.asarray([self.__process_path(path) for path in paths])\n        if self.labels:\n            y = np.array(list(zip(data_slice['L'].astype(int), data_slice['R'].astype(int))))\n            return X, y\n        else:\n            return X\n        \n    def __process_path(self, path):\n        img = tf.io.read_file(path)\n        img = tf.io.decode_png(img, channels=1)\n        img.set_shape([None, None, 1])\n        img = tf.image.convert_image_dtype(img, tf.float32)\n        img = tf.image.resize(img, (256, 256))\n        return img\n    \n    def on_epoch_end(self):\n        \"\"\"Updates indexes after each epoch\"\"\"\n        if self.shuffle:\n            self.df = self.df.sample(frac=1).reset_index(drop=True)\n    \ntrain_gen = MyDataSet(train_labels_df, IMAGE_PATH,\n                      batch_size=64, labels=True)\n\ntest_gen = MyDataSet(test_only_first, '/kaggle/working/output/',\n                      batch_size=32, labels=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:37:03.203377Z","iopub.execute_input":"2023-01-25T16:37:03.203823Z","iopub.status.idle":"2023-01-25T16:37:03.223563Z","shell.execute_reply.started":"2023-01-25T16:37:03.203769Z","shell.execute_reply":"2023-01-25T16:37:03.222563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from imgaug import augmenters as iaa\n\n\n\n# class DataGenerator(tf.keras.utils.Sequence):\n#     def __init__(self, df, path, batch_size=32, shuffle=True,aug=True,labels=True):\n#         self.df = df.copy()\n#         self.patient_ids = self.df['patient_id'].unique()\n#         self.labels = labels\n#         if self.labels ==True:\n# #             self.labels = self.df.groupby('prediction_id')['cancer'].max()\n#             self.labels = self.df[['L', 'R']]\n#         self.path = path\n#         self.batch_size = batch_size\n#         self.aug=aug\n#         self.shuffle = shuffle\n#         self.on_epoch_end()\n\n#     def __len__(self):\n#         \"\"\"Denotes the number of batches per epoch\"\"\"\n#         return int(len(self.patient_ids) / self.batch_size)\n\n#     def __getitem__(self, index):\n#         \"\"\"Generate one batch of data\"\"\"\n#         batch_indexes = self.prediction_ids[index * self.batch_size:(index + 1) * self.batch_size]\n#         X, y = self.__data_generation(batch_indexes)\n#         return X, y\n\n#     def __get_input(self, path):\n        \n# #         print('path   ',path)\n#         image = tf.keras.preprocessing.image.load_img(path)\n#         image_arr = tf.keras.preprocessing.image.img_to_array(image)\n        \n#         if self.aug:\n            \n#              image_arr=self.augmentor(image_arr)\n\n        \n#         return image_arr\n\n    \n#     def augmentor(self, images):\n#         'Apply data augmentation'\n#         images=data_augmentation_layers(images)\n\n#         return images\n    \n    \n#     def on_epoch_end(self):\n#         \"\"\"Updates indexes after each epoch\"\"\"\n#         if self.shuffle:\n#             self.df = self.df.sample(frac=1).reset_index(drop=True)\n\n#     def __data_generation(self, batch_indexes):\n#         paths = self.get_paths_images(batch_indexes)\n#         X = np.asarray([self.__get_input(path) for path in paths])\n#         y = np.array([self.labels[batch_indexes]])\n#         return X, y\n\n#     def get_paths_images(self, batch_indexes):\n#         batch = self.df[self.df['patient_id'].isin(batch_indexes)]\n#         rows_batch = self.get_rows(batch)\n#         return self.path + rows_batch[\"patient_id\"].astype(str) + \"/\" + rows_batch[\"image_id\"].astype(\n#             str) + \".png\"\n\n#     def get_rows(self, batch):\n#         \"\"\"Select only 1 MLO view picture per breast\"\"\"\n#         only_MLO_view_images = batch[batch['view'] == 'MLO']\n#         only_one_per_prediction_id = only_MLO_view_images.groupby('patient_id')[['patient_id', 'image_id']].max()\n#         return only_one_per_prediction_id\n    \n# train_gen = DataGenerator(train_labels_df, batch_size=BATCH_SIZE, path=IMAGE_PATH,aug=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-18T18:00:04.303221Z","iopub.execute_input":"2023-01-18T18:00:04.304114Z","iopub.status.idle":"2023-01-18T18:00:04.311328Z","shell.execute_reply.started":"2023-01-18T18:00:04.304058Z","shell.execute_reply":"2023-01-18T18:00:04.310148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_batch, label_batch = next(iter(train_gen))\n\n# plt.figure(figsize=(10, 10))\n# for i in range(9):\n#     ax = plt.subplot(3, 3, i+1)\n#     plt.imshow(image_batch[i].numpy())\n#     label = 'L'+str(int(label_batch['L'].numpy()[i])) +'_'+ 'R'+str(int(label_batch['R'].numpy()[i]))\n#     plt.title(label)","metadata":{"execution":{"iopub.status.busy":"2023-01-18T18:00:04.751033Z","iopub.execute_input":"2023-01-18T18:00:04.751724Z","iopub.status.idle":"2023-01-18T18:00:04.756568Z","shell.execute_reply.started":"2023-01-18T18:00:04.75169Z","shell.execute_reply":"2023-01-18T18:00:04.755394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications import MobileNet, InceptionV3, ResNet50\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense,GlobalAveragePooling2D, Dropout, Flatten\nfrom tensorflow.keras import layers\n# base_model=ResNet50(weights='imagenet',include_top=False, input_shape=(256, 256, 3))","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:37:09.830388Z","iopub.execute_input":"2023-01-25T16:37:09.830735Z","iopub.status.idle":"2023-01-25T16:37:09.837746Z","shell.execute_reply.started":"2023-01-25T16:37:09.830706Z","shell.execute_reply":"2023-01-25T16:37:09.836673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class myBlock(layers.Layer):\n    def __init__(self, out_channel, kernel_size=3):\n        super(myBlock, self).__init__()\n        self.conv = layers.Conv2D(out_channel, kernel_size, padding='same')\n        self.bn = layers.BatchNormalization()\n        \n    def call(self, input_tensor, training=False):\n        x = self.conv(input_tensor)\n        x = self.bn(x)\n        x = tf.nn.relu(x)\n        return x\n\n\nclass ResBlock(layers.Layer):\n    def __init__(self, channels):\n        super(ResBlock, self).__init__()\n        self.cnn1 = myBlock(channels[0])\n        self.cnn2 = myBlock(channels[1])\n        self.cnn3 = myBlock(channels[2])\n        self.pooling = layers.MaxPooling2D()\n        self.identity_mapping = layers.Conv2D(channels[1], 3, padding='same')\n        \n    def call(self, input_tensor, training=False):\n        x = self.cnn1(input_tensor, training=training)\n        x = self.cnn2(x, training=training)\n        x = self.cnn3(x + self.identity_mapping(input_tensor), training=training)\n        x = self.pooling(x)\n        return x\n    \nclass Resnet_Like(tf.keras.Model):\n    def __init__(self, num_classes):\n        super(Resnet_Like, self).__init__()\n        self.block1 = ResBlock([32, 32, 64])\n        self.block2 = ResBlock([128, 128, 256])\n        self.block3 = ResBlock([128, 256, 512])\n        self.pool = layers.GlobalAveragePooling2D()\n        self.classifier = layers.Dense(num_classes, activation='sigmoid')\n    \n    def call(self, input_tensor, training=False):\n        x = self.block1(input_tensor, training=False)\n        x = self.block2(x, training=False)\n        x = self.block3(x, training=False)\n        x = self.pool(x)\n        return self.classifier(x)\n        \nmodel = Resnet_Like(2)\n# model = tf.keras.Sequential([\n#     myBlock(64),\n#     myBlock(32),\n#     layers.Flatten(),\n#     layers.Dense(2, activation='sigmoid')\n# ])\n","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:37:12.227384Z","iopub.execute_input":"2023-01-25T16:37:12.227745Z","iopub.status.idle":"2023-01-25T16:37:14.30626Z","shell.execute_reply.started":"2023-01-25T16:37:12.227715Z","shell.execute_reply":"2023-01-25T16:37:14.305255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x=base_model.output\n# x=GlobalAveragePooling2D()(x)\n# x = Flatten()(x)\n# x=Dense(512,activation='relu')(x)\n# dropout1 = Dropout(0.70)\n# x = dropout1(x)\n# x=Dense(512,activation='relu')(x)\n# dropout2 = Dropout(0.30)\n# x = dropout2(x)\n# preds=Dense(2,activation='sigmoid')(x)\n\n# model=Model(inputs=base_model.input,outputs=preds)\n\n# freeze_point = 167\n# for layer in model.layers[:freeze_point]:\n#     layer.trainable=False\n# for layer in model.layers[freeze_point:]:\n#     layer.trainable=True","metadata":{"execution":{"iopub.status.busy":"2023-01-18T18:00:24.059948Z","iopub.execute_input":"2023-01-18T18:00:24.060659Z","iopub.status.idle":"2023-01-18T18:00:24.065463Z","shell.execute_reply.started":"2023-01-18T18:00:24.060619Z","shell.execute_reply":"2023-01-18T18:00:24.064346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adam = tf.optimizers.Adam(learning_rate=0.0001)\n# model.compile(optimizer=adam,loss='binary_crossentropy',metrics=['accuracy'])\nmodel.compile(optimizer=adam, loss='binary_crossentropy',metrics=tfa.metrics.FBetaScore(num_classes=2))","metadata":{"execution":{"iopub.status.busy":"2023-01-25T16:37:15.504955Z","iopub.execute_input":"2023-01-25T16:37:15.505318Z","iopub.status.idle":"2023-01-25T16:37:15.531632Z","shell.execute_reply.started":"2023-01-25T16:37:15.50529Z","shell.execute_reply":"2023-01-25T16:37:15.530694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# STEP_SIZE_TRAIN=train_generator.n//train_generator.batch_size\n# STEP_SIZE_VALID=valid_generator.n//valid_generator.batch_size\nstart = time.time()\n\nhistory = model.fit(\n        train_gen,\n        epochs=15,\n#         validation_data=valid_generator\n)\n\n# history = tpu_model.fit_generator(\n#         train_generator,\n#         epochs=20,\n#         steps_per_epoch=STEP_SIZE_TRAIN,\n#         validation_data=valid_generator,\n#         validation_steps=STEP_SIZE_VALID\n# )\n\nprint(\"Time taken to train: \", (time.time() - start)/60)","metadata":{"execution":{"iopub.status.busy":"2023-01-18T18:00:25.427128Z","iopub.execute_input":"2023-01-18T18:00:25.427545Z","iopub.status.idle":"2023-01-18T19:05:40.910644Z","shell.execute_reply.started":"2023-01-18T18:00:25.427511Z","shell.execute_reply":"2023-01-18T19:05:40.908781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_gen = MyDataSet(test_only_first, '/kaggle/working/output/', \n#                      batch_size=1, labels=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-18T18:00:08.840967Z","iopub.status.idle":"2023-01-18T18:00:08.842127Z","shell.execute_reply.started":"2023-01-18T18:00:08.841841Z","shell.execute_reply":"2023-01-18T18:00:08.841865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class MyFullDataSet(tf.keras.utils.Sequence):\n#     def __init__(self, df, shuffle=True, batch_size=32, labels=True):\n#         self.df = df\n#         self.patient_ids = self.df.patient_id.unique()\n#         if labels:\n#             self.labels = self.df[['L', 'R']]\n#         self.batch_size = batch_size\n#         self.shuffle = shuffle\n#         self.on_epoch_end()\n    \n#     def __len__(self):\n#         \"\"\"Denotes the number of batches per epoch\"\"\"\n#         return int(len(self.patient_ids) / self.batch_size)\n    \n#     def __getitem__(self, index):\n#         batch_indexes = self.patient_ids[index * self.batch_size:(index + 1) * self.batch_size]\n#         data_slice = self.df[self.df['patient_id'].isin(batch_indexes)]\n#         X = np.asarray([self.combine_patient_image(patient_id) for patient_id in batch_indexes])\n#         return X\n\n#     def combine_patient_image(self, patient_id):\n#         image_dict = {}\n#         dcm_paths = self.df[self.df['patient_id'] == patient_id]['image_path'].values\n#         for image_path in dcm_paths:\n#             image_id = str(image_path).split('/')[-1].split('.')[0]\n#             key = self.df[self.df['image_id'] == int(image_id)]['laterality_view'].values[0]\n#             dcm_img = dcm.dcmread(image_path).pixel_array\n#             w, h = dcm_img.shape\n\n#             img = np.empty((w, h, 3), dtype=dcm_img.dtype)\n\n#             img[:,:,2] = img[:,:,1] = img[:,:,0] = dcm_img\n# #             img = tf.image.convert_image_dtype(img, tf.float32)\n#             img = tf.image.resize(img, (256, 256))\n# #             img = cv2.imread(str(os.path.join(DATA_PATH, image_path)))\n#             breast_loc = self._check_breast_location(img)\n#             if key[0] != breast_loc:\n#                 img = cv2.flip(img, 1)\n#             image_dict[key] = img\n#         mlo = np.hstack((image_dict['R_MLO'], image_dict['L_MLO']))\n#         cc = np.hstack((image_dict['R_CC'], image_dict['L_CC']))\n#         t = np.vstack((mlo, cc))\n#     #     cv2.imwrite(\n#     #             f'/kaggle/working/output/{patient_id}.png',\n#     #             t\n#     #         )\n#         return t\n    \n#     def _check_breast_location(self, img):\n#         left_half = img[:, :img.shape[1]//2]\n#         right_half = img[:, img.shape[1]//2:]\n#         if np.mean(left_half) > np.mean(right_half):\n#             return \"L\"\n#         else:\n#             return \"R\"\n\n#     def on_epoch_end(self):\n#         \"\"\"Updates indexes after each epoch\"\"\"\n#         if self.shuffle:\n#             self.df = self.df.sample(frac=1).reset_index(drop=True)\n    \n# test_gen = MyFullDataSet(test_only_first, batch_size=1, labels=False, shuffle=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred =  model.predict(test_gen)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T21:01:09.831029Z","iopub.execute_input":"2023-01-16T21:01:09.831405Z","iopub.status.idle":"2023-01-16T21:01:09.902807Z","shell.execute_reply.started":"2023-01-16T21:01:09.831373Z","shell.execute_reply":"2023-01-16T21:01:09.901883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(pred, columns=['L', 'R'])\nsubmission['patient_id'] = test_gen.patient_ids\nsubmission = pd.melt(submission, id_vars=[\"patient_id\"], value_vars=[\"L\", \"R\"], var_name=\"laterality\", value_name=\"cancer\")\nsubmission['prediction_id'] = submission['patient_id'].astype(str) + '_' +submission['laterality']\nsubmission['cancer']=(submission.cancer > 0.5).astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T21:01:26.757606Z","iopub.execute_input":"2023-01-16T21:01:26.758087Z","iopub.status.idle":"2023-01-16T21:01:26.780556Z","shell.execute_reply.started":"2023-01-16T21:01:26.758042Z","shell.execute_reply":"2023-01-16T21:01:26.779627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = submission[['prediction_id', 'cancer']]\nsubmit = submit.fillna(0.0)\nsubmit.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-16T21:01:36.455581Z","iopub.execute_input":"2023-01-16T21:01:36.455988Z","iopub.status.idle":"2023-01-16T21:01:36.468231Z","shell.execute_reply.started":"2023-01-16T21:01:36.455931Z","shell.execute_reply":"2023-01-16T21:01:36.467127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission[['prediction_id', 'cancer']].to_csv('submission.csv', index=False)\nsubmit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-16T21:01:41.488759Z","iopub.execute_input":"2023-01-16T21:01:41.48914Z","iopub.status.idle":"2023-01-16T21:01:41.497235Z","shell.execute_reply.started":"2023-01-16T21:01:41.489105Z","shell.execute_reply":"2023-01-16T21:01:41.496161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}