{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n!pip install tf_clahe\n!pip install tensorflow-hub\n!pip install tensorflow-datasets\n!pip install tensorflow-io","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:17:07.86636Z","iopub.execute_input":"2023-01-22T14:17:07.867274Z","iopub.status.idle":"2023-01-22T14:18:26.062253Z","shell.execute_reply.started":"2023-01-22T14:17:07.867157Z","shell.execute_reply":"2023-01-22T14:18:26.060939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport glob\nimport sys\nimport time\nimport random\nimport pydicom\nimport tf_clahe\nimport numpy as np \nimport pandas as pd \nimport scipy.ndimage\nimport seaborn as sns\nfrom PIL import Image\n\nfrom matplotlib import rcParams\nimport matplotlib.pyplot as plt\n\n\nfrom os import listdir, mkdir\n\nimport tensorflow as tf\nfrom tensorflow import keras\nimport tensorflow_io as tfio\n\nfrom skimage.transform import resize \n\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.applications.vgg16 import VGG16\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, TensorBoard, ModelCheckpoint\n# Import Keras Libraries\n\nimport logging\nlogging.getLogger('tensorflow').disabled = True\nos.environ[\"TF_CPP_MIN_LOG_LEVEL\"] = '3'\nprint(f'Tensorflow Version: {tf.__version__}')\nprint(f'Python Version: {sys.version}')\n\n%load_ext tensorboard","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:23:21.523688Z","iopub.execute_input":"2023-01-22T14:23:21.524133Z","iopub.status.idle":"2023-01-22T14:23:21.537076Z","shell.execute_reply.started":"2023-01-22T14:23:21.524099Z","shell.execute_reply":"2023-01-22T14:23:21.535779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I will be using this notebook to optimize RSNA classification problem using a few preprocessing techniques and transfer learning after a very basic CNN model which you can find [here](https://www.kaggle.com/code/kuntalpal/rsna-baseline). Work in progress!!","metadata":{}},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() # TPU detection\nexcept ValueError:\n    tpu = None\n    gpus = tf.config.experimental.list_logical_devices(\"GPU\")\n    \nif tpu:\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    STRATEGY = tf.distribute.experimental.TPUStrategy(tpu,) \n    print('Running on TPU ', tpu.cluster_spec().as_dict()['worker'])\nelif len(gpus) > 1:\n    STRATEGY = tf.distribute.MirroredStrategy([gpu.name for gpu in gpus])\n    print('Running on multiple GPUs ', [gpu.name for gpu in gpus])\nelif len(gpus) == 1:\n    STRATEGY = tf.distribute.get_strategy() \n    print('Running on single GPU ', gpus[0].name)\nelse:\n    STRATEGY = tf.distribute.get_strategy() \n    print('Running on CPU')\nprint(\"Number of accelerators: \", STRATEGY.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:37.495185Z","iopub.execute_input":"2023-01-22T14:18:37.497586Z","iopub.status.idle":"2023-01-22T14:18:43.23735Z","shell.execute_reply.started":"2023-01-22T14:18:37.497543Z","shell.execute_reply":"2023-01-22T14:18:43.236172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DF_PATH = '/kaggle/input/rsna-breast-cancer-detection'\nIMG_PATH = '/kaggle/input/rsna-breast-cancer-512-pngs'\n\nBATCH_SIZE = 32\n    \ndf = pd.read_csv(f\"{DF_PATH}/train.csv\")\ndf['img_path'] = df.apply(\n    lambda i: os.path.join(\n        f\"{IMG_PATH}\", str(i['patient_id']) + \"_\" + str(i['image_id']) + '.png'\n    ), axis=1\n)\n\ndisplay(df.head())\nprint(df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:43.240074Z","iopub.execute_input":"2023-01-22T14:18:43.240709Z","iopub.status.idle":"2023-01-22T14:18:44.239708Z","shell.execute_reply.started":"2023-01-22T14:18:43.240669Z","shell.execute_reply":"2023-01-22T14:18:44.238645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create train-validation split\ntrain_df, valid_df = train_test_split(df[['img_path','cancer']], test_size=0.2, random_state=42)\n\n\nprint(train_df.cancer.value_counts(normalize=True))\nprint(valid_df.cancer.value_counts(normalize=True))","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:44.240969Z","iopub.execute_input":"2023-01-22T14:18:44.241864Z","iopub.status.idle":"2023-01-22T14:18:44.272415Z","shell.execute_reply.started":"2023-01-22T14:18:44.241823Z","shell.execute_reply":"2023-01-22T14:18:44.271117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\n\n# figure size in inches optional\nrcParams['figure.figsize'] = 11 ,8\n\n# display images\nfig, ax = plt.subplots(1,3)\nax[0].imshow(cv2.imread(df['img_path'][0]))\nax[1].imshow(cv2.imread(df['img_path'][1]))\nax[2].imshow(cv2.imread(df['img_path'][5]))","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:44.274247Z","iopub.execute_input":"2023-01-22T14:18:44.274958Z","iopub.status.idle":"2023-01-22T14:18:44.955007Z","shell.execute_reply.started":"2023-01-22T14:18:44.274902Z","shell.execute_reply":"2023-01-22T14:18:44.953909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Histogram Equalization\n\nHistogram equalization is a method in image processing of contrast adjustment using the image's histogram.\n\n\nThis method usually increases the global contrast of many images, especially when the image is represented by a narrow range of intensity values. Through this adjustment, the intensities can be better distributed on the histogram utilizing the full range of intensities evenly. This allows for areas of lower local contrast to gain a higher contrast. Histogram equalization accomplishes this by effectively spreading out the highly populated intensity values which are used to degrade image contrast.\n\n[source Wikipedia](https://en.wikipedia.org/wiki/Histogram_equalization)","metadata":{}},{"cell_type":"code","source":"img = cv2.imread(df['img_path'][0], cv2.IMREAD_GRAYSCALE)\nequ = cv2.equalizeHist(img)\nres = np.hstack((img,equ)) #stacking images side-by-side\nplt.imshow(res)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:44.956416Z","iopub.execute_input":"2023-01-22T14:18:44.957133Z","iopub.status.idle":"2023-01-22T14:18:45.294164Z","shell.execute_reply.started":"2023-01-22T14:18:44.957091Z","shell.execute_reply":"2023-01-22T14:18:45.292897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CLAHE (Contrast Limited Adaptive Histogram Equalization)\n\nIn adaptive histogram equalization, image is divided into small blocks called \"tiles\" (tileSize is 8x8 by default in OpenCV). Then each of these blocks are histogram equalized as usual. So in a small area, histogram would confine to a small region (unless there is noise). If noise is there, it will be amplified. To avoid this, contrast limiting is applied. If any histogram bin is above the specified contrast limit (by default 40 in OpenCV), those pixels are clipped and distributed uniformly to other bins before applying histogram equalization. After equalization, to remove artifacts in tile borders, bilinear interpolation is applied.","metadata":{}},{"cell_type":"code","source":"img = tf.io.read_file(df['img_path'][1])\nimg = tf.image.decode_png(img, channels=3)\n# For ease of understanding, we explicitly equalize each channel individually\ncla_r = tf_clahe.clahe(img[:,:,0],tile_grid_size=(8, 8), clip_limit=2.0)\ncla_g = tf_clahe.clahe(img[:,:,1],tile_grid_size=(8, 8), clip_limit=2.0)\ncla_b = tf_clahe.clahe(img[:,:,2],tile_grid_size=(8, 8), clip_limit=2.0)\n\n# Next we stack our equalized channels back into a single image\ncla = np.stack((cla_r,cla_g,cla_b), axis=2)\n\nres = np.hstack((img,cla)) #stacking images side-by-side\nplt.imshow(res)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:45.295563Z","iopub.execute_input":"2023-01-22T14:18:45.295918Z","iopub.status.idle":"2023-01-22T14:18:48.014857Z","shell.execute_reply.started":"2023-01-22T14:18:45.295886Z","shell.execute_reply":"2023-01-22T14:18:48.013908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@tf.function(experimental_compile=True)  # Enable XLA\ndef clahe_eq(image):\n    # For ease of understanding, we explicitly equalize each channel individually\n    cla_r = tf_clahe.clahe(image[:,:,0],tile_grid_size=(8, 8), clip_limit=2.0, gpu_optimized=True)\n    cla_g = tf_clahe.clahe(image[:,:,1],tile_grid_size=(8, 8), clip_limit=2.0, gpu_optimized=True)\n    cla_b = tf_clahe.clahe(image[:,:,2],tile_grid_size=(8, 8), clip_limit=2.0, gpu_optimized=True)\n\n    # Next we stack our equalized channels back into a single image\n    cla = tf.stack((cla_r,cla_g,cla_b), axis=2)\n    \n    return cla","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:48.016523Z","iopub.execute_input":"2023-01-22T14:18:48.017279Z","iopub.status.idle":"2023-01-22T14:18:48.024831Z","shell.execute_reply.started":"2023-01-22T14:18:48.017234Z","shell.execute_reply":"2023-01-22T14:18:48.023767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_image(file_path, labels, apply_clahe=True, apply_hist_eq=False):\n    \n    image = tf.io.read_file(file_path)\n    image = tf.image.decode_png(image, channels=3)\n    \n    \n    # Apply histogram equalization\n    if apply_hist_eq:\n        \n        image = cv2.equalizeHist(image)\n    \n    # Apply CLAHE contrast enhancement\n    if apply_clahe:\n        image = clahe_eq(image)\n    \n    return image, labels","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:48.030415Z","iopub.execute_input":"2023-01-22T14:18:48.03069Z","iopub.status.idle":"2023-01-22T14:18:48.044014Z","shell.execute_reply.started":"2023-01-22T14:18:48.030665Z","shell.execute_reply":"2023-01-22T14:18:48.043007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_dataset(df,batch_size,with_labels = False, shuffle= False):\n    \n    \"\"\"\n    Generates a TF dataset for feeding in the data.\n    Originally written as a group for the common pipeline.\n    :param x: X inputs - paths to images\n    :param y: y values - labels for images\n    :return: the dataset\n    \"\"\"\n    \n    # Create Dataset\n    if with_labels:\n        dataset = tf.data.Dataset.from_tensor_slices((df['img_path'].values, df['cancer'].values))\n    else:\n        dataset = tf.data.Dataset.from_tensor_slices(\n            (df['img_path'].values)\n        )\n\n    # Image preprocessing\n    dataset = dataset.map(process_image) \n    #dataset.map(lambda filename: tuple(tf.py_function(image_process, [filename], [tf.uint8])))\n\n    # Get one batch\n    dataset = dataset.shuffle(1024, reshuffle_each_iteration = True) if shuffle else dataset\n    dataset = dataset.batch(batch_size,drop_remainder=False)\n    dataset = dataset.prefetch(tf.data.AUTOTUNE)\n    #dataset = dataset.shape\n\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:48.04554Z","iopub.execute_input":"2023-01-22T14:18:48.04603Z","iopub.status.idle":"2023-01-22T14:18:48.059031Z","shell.execute_reply.started":"2023-01-22T14:18:48.045993Z","shell.execute_reply":"2023-01-22T14:18:48.057948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_dataset = create_dataset(\n    train_df,\n    batch_size  = BATCH_SIZE, \n    with_labels = True, \n    shuffle = True\n)\n\nvalid_dataset = create_dataset(\n    valid_df,\n    batch_size  = BATCH_SIZE, \n    with_labels = True, \n    shuffle = False\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:48.060976Z","iopub.execute_input":"2023-01-22T14:18:48.061402Z","iopub.status.idle":"2023-01-22T14:18:48.472936Z","shell.execute_reply.started":"2023-01-22T14:18:48.061365Z","shell.execute_reply":"2023-01-22T14:18:48.472007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tf_pfbeta(from_logits=True, beta=1.0, epsilon=1e-07):\n    \n    def pfbeta(y_true, y_pred):\n        y_pred = tf.cond(\n            tf.cast(from_logits, dtype=tf.bool),\n            lambda: tf.nn.sigmoid(y_pred),\n            lambda: y_pred,\n        )\n        y_true = tf.reshape(y_true, [-1])\n        y_pred = tf.reshape(y_pred, [-1])\n\n        ctp = tf.reduce_sum(y_true * y_pred, axis=-1)\n        cfp = tf.reduce_sum(y_pred, axis=-1) - ctp\n\n        c_precision = ctp / (ctp + cfp)\n        c_recall = ctp / tf.reduce_sum(y_true)\n        \n        def compute_fractions():\n            numerator = c_precision * c_recall\n            denominator = beta**2 * c_precision + c_recall\n            return (1 + beta**2) * tf.math.divide_no_nan(numerator, denominator)\n        \n        return tf.cond(\n            tf.logical_and(\n                tf.greater(c_precision, 0.), tf.greater(c_recall, 0.)\n            ),\n            compute_fractions,\n            lambda: tf.constant(0, dtype=tf.float32)\n        )\n    \n    return pfbeta","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:48.474238Z","iopub.execute_input":"2023-01-22T14:18:48.474574Z","iopub.status.idle":"2023-01-22T14:18:48.485345Z","shell.execute_reply.started":"2023-01-22T14:18:48.47454Z","shell.execute_reply":"2023-01-22T14:18:48.484023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def binary_focal_loss(gamma=2., alpha=.25, class_weights=None):\n    \"\"\"\n    Binary form of focal loss.\n      FL(p_t) = -alpha * (1 - p_t)**gamma * log(p_t)\n      where p = sigmoid(x), p_t = p or 1 - p depending on if the label is 1 or 0, respectively.\n    References:\n        https://arxiv.org/pdf/1708.02002.pdf\n    Usage:\n     model.compile(loss=[binary_focal_loss(alpha=.25, gamma=2)], metrics=[\"accuracy\"], optimizer=adam)\n    \"\"\"\n    def binary_focal_loss_fixed(y_true, y_pred):\n        \"\"\"\n        y_true shape need be (None,1)\n        y_pred need be compute after sigmoid\n        \"\"\"\n        y_true = tf.cast(y_true, tf.float32)\n        y_pred = tf.clip_by_value(y_pred, tf.keras.backend.epsilon(), 1-tf.keras.backend.epsilon())\n        pt_1 = tf.where(tf.equal(y_true, 1), y_pred, tf.ones_like(y_pred))\n        pt_0 = tf.where(tf.equal(y_true, 0), y_pred, tf.zeros_like(y_pred))\n\n        epsilon = tf.keras.backend.epsilon()\n        # clip to prevent NaN's and Inf's\n        pt_1 = tf.keras.backend.clip(pt_1, epsilon, 1. - epsilon)\n        pt_0 = tf.keras.backend.clip(pt_0, epsilon, 1. - epsilon)\n        \n        if class_weights is not None:\n            weight_0 = class_weights[0]\n            weight_1 = class_weights[1]\n            \n            loss = -weight_0*tf.keras.backend.mean(alpha * tf.keras.backend.pow(1. - pt_1, gamma) * tf.keras.backend.log(pt_1)) \\\n               -weight_1*tf.keras.backend.mean((1 - alpha) * tf.keras.backend.pow(pt_0, gamma) * tf.keras.backend.log(1. - pt_0))\n            \n        else:\n            \n            loss = -tf.keras.backend.mean(alpha * tf.keras.backend.pow(1. - pt_1, gamma) * tf.keras.backend.log(pt_1)) \\\n               -tf.keras.backend.mean((1 - alpha) * tf.keras.backend.pow(pt_0, gamma) * tf.keras.backend.log(1. - pt_0))\n        \n        \n        return loss\n\n    return binary_focal_loss_fixed","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:48.487276Z","iopub.execute_input":"2023-01-22T14:18:48.488084Z","iopub.status.idle":"2023-01-22T14:18:48.501023Z","shell.execute_reply.started":"2023-01-22T14:18:48.488043Z","shell.execute_reply":"2023-01-22T14:18:48.500079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_augmentation_layers = tf.keras.Sequential(\n    [\n        tf.layers.experimental.preprocessing.RandomCrop(height=512, width=512),\n        tf.layers.experimental.preprocessing.RandomFlip(\"horizontal_and_vertical\"),\n        tf.layers.experimental.preprocessing.RandomRotation(0.25),\n        tf.layers.experimental.preprocessing.RandomZoom((-0.2, 0)),\n        tf.layers.experimental.preprocessing.RandomContrast((0.2,0.2)),\n])","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:18:48.502846Z","iopub.execute_input":"2023-01-22T14:18:48.503252Z","iopub.status.idle":"2023-01-22T14:18:48.833196Z","shell.execute_reply.started":"2023-01-22T14:18:48.503207Z","shell.execute_reply":"2023-01-22T14:18:48.831789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n    # Create the model\n    # Verify Mixed Policy Settings\n    print(f'Compute dtype: {tf.keras.mixed_precision.global_policy().compute_dtype}')\n    print(f'Variable dtype: {tf.keras.mixed_precision.global_policy().variable_dtype}')\n    \n    with STRATEGY.scope():\n        \n        base_model = VGG16(input_shape = (512, 512, 3), # Shape of our images\n        include_top = False, # Leave out the last fully connected layer\n        weights = 'imagenet')\n\n        base_model.trainable=False\n        \n        model = base_model.output\n\n        model = tf.keras.layers.Dense(512,activation='relu')(model)\n        model = tf.keras.layers.GlobalAveragePooling2D()(model)\n        model = tf.keras.layers.Dropout(rate=0.5)(model)\n        model = tf.keras.layers.Dense(1,activation='sigmoid')(model)\n        model = tf.keras.models.Model(inputs=base_model.input, outputs = model)\n        # We will use the famous Adam optimizer for fast learning\n        optimizer = tf.optimizers.Adam(learning_rate=0.001, epsilon=1e-7, clipnorm=10.0)\n        \n        # Loss\n        loss = binary_focal_loss(gamma=2., alpha=.25, class_weights = {0: 1., 1: 10.})\n\n        \n        # Metrics\n        metrics = [\n            #tfa.metrics.F1Score(num_classes=1, threshold=0.50),\n            tf_pfbeta(beta=1.0, from_logits=True),\n            tf.keras.metrics.Precision(),\n            tf.keras.metrics.Recall(),\n            tf.keras.metrics.AUC(),\n            tf.keras.metrics.BinaryAccuracy(),\n        ]\n        \n        model.compile(optimizer=optimizer, loss=loss, metrics=metrics)\n\n\n        return model","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:22:47.803055Z","iopub.execute_input":"2023-01-22T14:22:47.804073Z","iopub.status.idle":"2023-01-22T14:22:47.814996Z","shell.execute_reply.started":"2023-01-22T14:22:47.804035Z","shell.execute_reply":"2023-01-22T14:22:47.813854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pretrained File Path: '/kaggle/input/sartorius-training-dataset/model.h5'\ntf.keras.backend.clear_session()\n# enable XLA optmizations\ntf.config.optimizer.set_jit(True)\n\nmodel = create_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:22:48.557512Z","iopub.execute_input":"2023-01-22T14:22:48.558672Z","iopub.status.idle":"2023-01-22T14:22:49.400716Z","shell.execute_reply.started":"2023-01-22T14:22:48.558614Z","shell.execute_reply":"2023-01-22T14:22:49.398703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tensorboard1 = TensorBoard(log_dir = 'logs')\n\ncheckpoint1 = ModelCheckpoint(\"vgg16\",monitor=\"val_pfbeta\",save_best_only=True,mode=\"auto\",verbose=1)\nreduce_lr1 = ReduceLROnPlateau(monitor = 'val_pfbeta', factor = 0.4, patience = 2, min_delta = 0.0001,\n                              mode='auto',verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-01-22T14:23:29.37151Z","iopub.execute_input":"2023-01-22T14:23:29.371902Z","iopub.status.idle":"2023-01-22T14:23:30.094978Z","shell.execute_reply.started":"2023-01-22T14:23:29.371869Z","shell.execute_reply":"2023-01-22T14:23:30.09399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n        training_dataset,\n        validation_data = valid_dataset,\n        epochs = 10,\n        callbacks=[tensorboard1,checkpoint1,reduce_lr1],\n        verbose = 1,\n    );","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}