{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install tensorflow==2.2.0  --quiet\n!pip install tf-nightly --quiet","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:49:07.903884Z","iopub.execute_input":"2021-12-15T09:49:07.904277Z","iopub.status.idle":"2021-12-15T09:51:55.376222Z","shell.execute_reply.started":"2021-12-15T09:49:07.904218Z","shell.execute_reply":"2021-12-15T09:51:55.375219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport PIL\nimport time\nimport math\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import load_img\n\nSEED = 1337\nprint('Tensorflow version : {}'.format(tf.__version__))\n\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept ValueError:\n    strategy = tf.distribute.get_strategy() # for CPU and single GPU\n    print('Number of replicas:', strategy.num_replicas_in_sync)\n    \nprint(tf.__version__)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-15T09:47:57.2832Z","iopub.execute_input":"2021-12-15T09:47:57.283513Z","iopub.status.idle":"2021-12-15T09:48:01.57938Z","shell.execute_reply.started":"2021-12-15T09:47:57.283464Z","shell.execute_reply":"2021-12-15T09:48:01.578345Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data Loading\n","metadata":{}},{"cell_type":"code","source":"MAIN_DIR = '../input/prostate-cancer-grade-assessment'\nTRAIN_IMG_DIR = '../input/panda-tiles/train'\nTRAIN_MASKS_DIR = '../input/panda-tiles/masks'\ntrain_csv = pd.read_csv(os.path.join(MAIN_DIR, 'train.csv')) ","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:52:19.355937Z","iopub.execute_input":"2021-12-15T09:52:19.35631Z","iopub.status.idle":"2021-12-15T09:52:19.388312Z","shell.execute_reply.started":"2021-12-15T09:52:19.35625Z","shell.execute_reply":"2021-12-15T09:52:19.387535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_images = tf.io.gfile.glob(TRAIN_IMG_DIR + '/*_0.png')\nimg_ids = train_csv['image_id']","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:52:21.406443Z","iopub.execute_input":"2021-12-15T09:52:21.406797Z","iopub.status.idle":"2021-12-15T09:52:50.305429Z","shell.execute_reply.started":"2021-12-15T09:52:21.406742Z","shell.execute_reply":"2021-12-15T09:52:50.304281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for img_id in img_ids:\n    file_name = TRAIN_IMG_DIR + '/' + img_id + '_0.png'\n    if file_name not in valid_images:\n        train_csv = train_csv[train_csv['image_id'] != img_id]\n        \nradboud_csv = train_csv[train_csv['data_provider'] == 'radboud']\nkarolinska_csv = train_csv[train_csv['data_provider'] != 'radboud']","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:53:11.761884Z","iopub.execute_input":"2021-12-15T09:53:11.762253Z","iopub.status.idle":"2021-12-15T09:53:13.333106Z","shell.execute_reply.started":"2021-12-15T09:53:11.762196Z","shell.execute_reply":"2021-12-15T09:53:13.332199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r_train, r_test = train_test_split(\n    radboud_csv,\n    test_size=0.2, random_state=SEED\n)\n\nk_train, k_test = train_test_split(\n    karolinska_csv,\n    test_size=0.2, random_state=SEED\n)","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:53:16.997538Z","iopub.execute_input":"2021-12-15T09:53:16.997881Z","iopub.status.idle":"2021-12-15T09:53:17.013352Z","shell.execute_reply.started":"2021-12-15T09:53:16.99783Z","shell.execute_reply":"2021-12-15T09:53:17.012564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Concatenate the dataframes from the two different providers and we have our training dataset and our validation dataset.","metadata":{}},{"cell_type":"code","source":"train_df = pd.concat([r_train, k_train])\nvalid_df = pd.concat([r_test, k_test])\n\nprint(train_df.shape)\nprint(valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:53:21.465568Z","iopub.execute_input":"2021-12-15T09:53:21.465905Z","iopub.status.idle":"2021-12-15T09:53:21.477258Z","shell.execute_reply.started":"2021-12-15T09:53:21.465856Z","shell.execute_reply":"2021-12-15T09:53:21.475206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_DIM = (1536, 128)\nCLASSES_NUM = 6\nBATCH_SIZE = 32\nEPOCHS = 100\nN=12\n\nLEARNING_RATE = 1e-4\nFOLDED_NUM_TRAIN_IMAGES = train_df.shape[0]\nFOLDED_NUM_VALID_IMAGES = valid_df.shape[0]\nSTEPS_PER_EPOCH = FOLDED_NUM_TRAIN_IMAGES // BATCH_SIZE\nVALIDATION_STEPS = FOLDED_NUM_VALID_IMAGES // BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:53:29.235723Z","iopub.execute_input":"2021-12-15T09:53:29.236342Z","iopub.status.idle":"2021-12-15T09:53:29.242997Z","shell.execute_reply.started":"2021-12-15T09:53:29.236217Z","shell.execute_reply":"2021-12-15T09:53:29.242235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataGenerator(tf.keras.utils.Sequence):\n    \n    def __init__(self,\n                 image_shape,\n                 batch_size, \n                 df,\n                 img_dir,\n                 mask_dir,\n                 is_training=True\n                 ):\n        \n        self.image_shape = image_shape\n        self.batch_size = batch_size\n        self.df = df\n        self.img_dir = img_dir\n        self.mask_dir = mask_dir\n        self.is_training = is_training\n        self.indices = range(df.shape[0])\n        \n    def __len__(self):\n        return self.df.shape[0] // self.batch_size\n    \n    def on_epoch_start(self):\n        if self.is_training:\n            np.random.shuffle(self.indices)\n    \n    def __getitem__(self, index):\n        batch_indices = self.indices[index * self.batch_size : (index+1) * self.batch_size]\n        image_ids = self.df['image_id'].iloc[batch_indices].values\n        batch_images = [self.__getimages__(image_id) for image_id in image_ids]\n        batch_labels = [self.df[self.df['image_id'] == image_id]['isup_grade'].values[0] for image_id in image_ids]\n        batch_labels = tf.one_hot(batch_labels, CLASSES_NUM)\n        \n        return np.squeeze(np.stack(batch_images).reshape(-1, 1536, 128, 3)), np.stack(batch_labels)\n        \n    def __getimages__(self, image_id):\n        fnames = [image_id+'_'+str(i)+'.png' for i in range(N)]\n        images = []\n        for fn in fnames:\n            img = np.array(PIL.Image.open(os.path.join(self.img_dir, fn)).convert('RGB'))[:, :, ::-1]\n            images.append(img)\n        result = np.stack(images).reshape(1, 1536, 128, 3) / 255.0\n        return result","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:53:33.590808Z","iopub.execute_input":"2021-12-15T09:53:33.591429Z","iopub.status.idle":"2021-12-15T09:53:33.607041Z","shell.execute_reply.started":"2021-12-15T09:53:33.591204Z","shell.execute_reply":"2021-12-15T09:53:33.606008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will use the DataGenerator to create a generator for our training dataset and for our validation dataset. At each iteration of the generator, the generator will return a batch of images.","metadata":{}},{"cell_type":"code","source":"train_generator = DataGenerator(image_shape=IMG_DIM,\n                                batch_size=BATCH_SIZE,\n                                df=train_df,\n                                img_dir=TRAIN_IMG_DIR,\n                                mask_dir=TRAIN_MASKS_DIR)\n\nvalid_generator = DataGenerator(image_shape=IMG_DIM,\n                                batch_size=BATCH_SIZE,\n                                df=valid_df,\n                                img_dir=TRAIN_IMG_DIR,\n                                mask_dir=TRAIN_MASKS_DIR)","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:53:38.731093Z","iopub.execute_input":"2021-12-15T09:53:38.731481Z","iopub.status.idle":"2021-12-15T09:53:38.737078Z","shell.execute_reply.started":"2021-12-15T09:53:38.731428Z","shell.execute_reply":"2021-12-15T09:53:38.736118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualize","metadata":{}},{"cell_type":"code","source":"def show_tiles(image_batch, label_batch):\n    plt.figure(figsize=(20,20))\n    for n in range(10):\n        ax = plt.subplot(1,10,n+1)\n        plt.imshow(image_batch[n])\n        decoded = np.argmax(label_batch[n])\n        plt.title(decoded)\n        plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:53:41.433078Z","iopub.execute_input":"2021-12-15T09:53:41.433472Z","iopub.status.idle":"2021-12-15T09:53:41.44004Z","shell.execute_reply.started":"2021-12-15T09:53:41.43342Z","shell.execute_reply":"2021-12-15T09:53:41.439109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_batch, label_batch = next(iter(train_generator))","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:53:46.553083Z","iopub.execute_input":"2021-12-15T09:53:46.553456Z","iopub.status.idle":"2021-12-15T09:53:51.476776Z","shell.execute_reply.started":"2021-12-15T09:53:46.553405Z","shell.execute_reply":"2021-12-15T09:53:51.475788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The following 12 tiles were from a single image but has been converted to 12 tiles to reduce white space. We see that only the sections that led to the ISUP grade has been preserved.","metadata":{}},{"cell_type":"code","source":"show_tiles(image_batch, label_batch)","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:53:54.138622Z","iopub.execute_input":"2021-12-15T09:53:54.139187Z","iopub.status.idle":"2021-12-15T09:53:55.453467Z","shell.execute_reply.started":"2021-12-15T09:53:54.13911Z","shell.execute_reply":"2021-12-15T09:53:55.452317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_model():\n    data_augmentation = tf.keras.Sequential([\n        tf.keras.layers.experimental.preprocessing.RandomContrast(0.15, seed=SEED),\n        tf.keras.layers.experimental.preprocessing.RandomFlip(\"horizontal\", seed=SEED),\n        tf.keras.layers.experimental.preprocessing.RandomFlip(\"vertical\", seed=SEED),\n        tf.keras.layers.experimental.preprocessing.RandomTranslation(0.1, 0.1, seed=SEED)\n    ])\n    \n    base_model = tf.keras.applications.VGG16(input_shape=(*IMG_DIM, 3),\n                                             include_top=False,\n                                             weights='imagenet')\n    \n    base_model.trainable = True\n    \n    model = tf.keras.Sequential([\n        data_augmentation,\n        \n        base_model,\n        \n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(16, activation='relu'),\n        tf.keras.layers.BatchNormalization(),\n        tf.keras.layers.Dense(CLASSES_NUM, activation='softmax'),\n    ])\n    \n    model.compile(optimizer=tf.keras.optimizers.RMSprop(),\n                  loss='categorical_crossentropy',\n                  metrics=tf.keras.metrics.AUC(name='auc'))\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:54:19.896602Z","iopub.execute_input":"2021-12-15T09:54:19.897087Z","iopub.status.idle":"2021-12-15T09:54:19.910408Z","shell.execute_reply.started":"2021-12-15T09:54:19.897024Z","shell.execute_reply":"2021-12-15T09:54:19.90969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model = make_model()","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:54:24.724037Z","iopub.execute_input":"2021-12-15T09:54:24.724417Z","iopub.status.idle":"2021-12-15T09:54:24.763093Z","shell.execute_reply.started":"2021-12-15T09:54:24.72436Z","shell.execute_reply":"2021-12-15T09:54:24.761146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" Training the model\n","metadata":{}},{"cell_type":"code","source":"def exponential_decay(lr0, s):\n    def exponential_decay_fn(epoch):\n        return lr0 * 0.1 **(epoch / s)\n    return exponential_decay_fn\n\nexponential_decay_fn = exponential_decay(0.01, 20)\n\nlr_scheduler = tf.keras.callbacks.LearningRateScheduler(exponential_decay_fn)\n\ncheckpoint_cb = tf.keras.callbacks.ModelCheckpoint(\"panda_model.h5\",\n                                                    save_best_only=True)\n\nearly_stopping_cb = tf.keras.callbacks.EarlyStopping(patience=10,\n                                                     restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:54:41.275698Z","iopub.execute_input":"2021-12-15T09:54:41.276055Z","iopub.status.idle":"2021-12-15T09:54:41.282572Z","shell.execute_reply.started":"2021-12-15T09:54:41.275998Z","shell.execute_reply":"2021-12-15T09:54:41.281735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_generator, epochs=EPOCHS,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    validation_data=valid_generator,\n    validation_steps=VALIDATION_STEPS,\n    callbacks=[checkpoint_cb, early_stopping_cb, lr_scheduler]\n)","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:54:44.663685Z","iopub.execute_input":"2021-12-15T09:54:44.664032Z","iopub.status.idle":"2021-12-15T09:54:44.680584Z","shell.execute_reply.started":"2021-12-15T09:54:44.663978Z","shell.execute_reply":"2021-12-15T09:54:44.679119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/prostate-cancer-grade-assessment/train.csv')\n","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:58:22.466573Z","iopub.execute_input":"2021-12-15T09:58:22.466906Z","iopub.status.idle":"2021-12-15T09:58:22.484066Z","shell.execute_reply.started":"2021-12-15T09:58:22.466857Z","shell.execute_reply":"2021-12-15T09:58:22.48336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:58:32.791137Z","iopub.execute_input":"2021-12-15T09:58:32.791538Z","iopub.status.idle":"2021-12-15T09:58:32.809246Z","shell.execute_reply.started":"2021-12-15T09:58:32.791486Z","shell.execute_reply":"2021-12-15T09:58:32.808096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len_df = len(train_df)\nprint(f\"There are {len_df} train images\")","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:58:42.892108Z","iopub.execute_input":"2021-12-15T09:58:42.892484Z","iopub.status.idle":"2021-12-15T09:58:42.898881Z","shell.execute_reply.started":"2021-12-15T09:58:42.892429Z","shell.execute_reply":"2021-12-15T09:58:42.897743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['isup_grade'].hist(figsize = (10, 5))","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:58:52.989808Z","iopub.execute_input":"2021-12-15T09:58:52.990144Z","iopub.status.idle":"2021-12-15T09:58:53.156592Z","shell.execute_reply.started":"2021-12-15T09:58:52.990091Z","shell.execute_reply":"2021-12-15T09:58:53.155669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[['primary Gleason', 'secondary Gleason']] = train_df.gleason_score.str.split('+',expand=True)","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:59:02.675779Z","iopub.execute_input":"2021-12-15T09:59:02.676132Z","iopub.status.idle":"2021-12-15T09:59:02.704715Z","shell.execute_reply.started":"2021-12-15T09:59:02.676079Z","shell.execute_reply":"2021-12-15T09:59:02.703784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[['primary Gleason', 'secondary Gleason']]","metadata":{"execution":{"iopub.status.busy":"2021-12-15T09:59:15.235834Z","iopub.execute_input":"2021-12-15T09:59:15.236199Z","iopub.status.idle":"2021-12-15T09:59:15.254718Z","shell.execute_reply.started":"2021-12-15T09:59:15.236125Z","shell.execute_reply":"2021-12-15T09:59:15.253865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['primary Gleason'].hist(figsize = (10, 5))\ntrain_df['secondary Gleason'].hist(figsize = (10, 5))","metadata":{"execution":{"iopub.status.busy":"2021-12-15T10:00:13.109394Z","iopub.execute_input":"2021-12-15T10:00:13.109762Z","iopub.status.idle":"2021-12-15T10:00:13.299154Z","shell.execute_reply.started":"2021-12-15T10:00:13.109701Z","shell.execute_reply":"2021-12-15T10:00:13.297644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here, we see that the Gleason pattern 3 is the most common.","metadata":{}},{"cell_type":"code","source":"import openslide\nimport os\nfrom PIL import Image\nimport pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2021-12-15T10:12:32.084074Z","iopub.execute_input":"2021-12-15T10:12:32.084451Z","iopub.status.idle":"2021-12-15T10:12:32.089887Z","shell.execute_reply.started":"2021-12-15T10:12:32.0844Z","shell.execute_reply":"2021-12-15T10:12:32.088913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ex_img_path = '../input/prostate-cancer-grade-assessment/train_images/'+train_df['image_id'][np.random.choice(len(train_df))]+'.tiff'\nexample_image = openslide.OpenSlide(ex_img_path)","metadata":{"execution":{"iopub.status.busy":"2021-12-15T10:12:40.891055Z","iopub.execute_input":"2021-12-15T10:12:40.891455Z","iopub.status.idle":"2021-12-15T10:12:40.965036Z","shell.execute_reply.started":"2021-12-15T10:12:40.891399Z","shell.execute_reply":"2021-12-15T10:12:40.964274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = example_image.read_region(location=(0,0),level=2,size=(example_image.level_dimensions[2][0],example_image.level_dimensions[2][1]))\nprint(img.size)\nimg","metadata":{"execution":{"iopub.status.busy":"2021-12-15T10:13:13.631837Z","iopub.execute_input":"2021-12-15T10:13:13.632381Z","iopub.status.idle":"2021-12-15T10:13:13.822451Z","shell.execute_reply.started":"2021-12-15T10:13:13.632141Z","shell.execute_reply":"2021-12-15T10:13:13.821595Z"},"trusted":true},"execution_count":null,"outputs":[]}]}