{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"print(\"\\n... IMPORTS STARTING ...\\n\")\n\nprint(\"\\n\\tVERSION INFORMATION\")\n# Machine Learning and Data Science Imports\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport tensorflow_hub as tfhub; print(f\"\\t\\t– TENSORFLOW HUB VERSION: {tfhub.__version__}\");\nimport tensorflow_addons as tfa; print(f\"\\t\\t– TENSORFLOW ADDONS VERSION: {tfa.__version__}\");\nimport tensorflow_io as tfio; print(f\"\\t\\t– TENSORFLOW I/O VERSION: {tfio.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None;\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\");\nfrom sklearn.preprocessing import RobustScaler, PolynomialFeatures\nfrom pandarallel import pandarallel; pandarallel.initialize();\nfrom sklearn.model_selection import GroupKFold, StratifiedKFold\nfrom scipy.spatial import cKDTree\n\n# # RAPIDS\n# import cudf, cupy, cuml\n# from cuml.neighbors import NearestNeighbors\n# from cuml.manifold import TSNE, UMAP\n\n# Built In Imports\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom glob import glob\nimport openslide\nimport warnings\nimport requests\nimport hashlib\nimport imageio\nimport IPython\nimport sklearn\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports\nfrom matplotlib.colors import ListedColormap\nfrom matplotlib.patches import Rectangle\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.express as px\nimport tifffile as tif\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance; Image.MAX_IMAGE_PIXELS = 5_000_000_000;\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom matplotlib import animation, rc; rc('animation', html='jshtml')\nimport plotly\nimport PIL\nimport cv2\n\nimport plotly.io as pio\nprint(pio.renderers)\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\n    \nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def flatten_l_o_l(nested_list):\n    \"\"\" Flatten a list of lists \"\"\"\n    return [item for sublist in nested_list for item in sublist]\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pil_open_downscaled_slide(img_path, downsample_by=8, interpolation=None, reducing_gap=3.0, as_numpy=True):\n    \"\"\"\n    \n    Helper function to convert WSI into smaller downscaled version using PIL\n    \n    Timing details for MAYO CLINIC STRIP AI dataset:\n        SMALLEST IMAGE BY AREA (4417, 5314)\n            * Function takes ~0.5 seconds to run\n        MEDIAN IMAGE BY AREA   (17573, 38743)\n            * Function takes ~16 seconds to run\n        LARGEST IMAGE BY AREA  (48282, 101406)(~208X LARGER THAN SMALLEST IMAGE)(~7X LARGER THAN MEDIAN IMAGE)\n            * Function will CRASH!\n    \n    Args:\n        img_path (str): Path to .tif file to be downsampled\n        downsample_by (int): How many times smaller should resultant \n            image be. i.e. image_size*(1/downsample_by) = new_size\n        interpolation (int, optional): The value that enumerates\n            the type of interpolation to be used in downsampling.\n            This can be one\n               * `Image.NEAREST`  --> ENUM=0\n               * `Image.BOX`      --> ENUM=1\n               * `Image.BILINEAR` --> ENUM=2\n               * `Image.HAMMING`  --> ENUM=3\n               * `Image.BICUBIC`  --> ENUM=4\n               * `Image.LANCZOS`  --> ENUM=5\n           If omitted, it defaults to `Image.BILINEAR`\n       reducing_gap (float, optional): How many steps to use when interpolating/resampling\n           the image after integer downscaling is performed. \n           ELI5: \n               * Higher (Approaching 5) means it takes longer but is better quality\n               * Lower  (Approaching 0) means it is much faster but  lower  quality\n        as_numpy (bool, optional): Whether to return image as numpy array (default)\n           or leave as PIL.Image object for further manipulation\n    \n    Returns:\n        Downsampled image as a numpy array of type uint8 with only 3 channels\n    \"\"\"\n    \n    # Catches and warning\n    if downsample_by<8: print(\"\\n... WARNING – DUE TO LOW DOWNSCALE_BY VALUE THE RETURND IMAGE MAY OCCUPY A LARGE AMOUNT OF RAM – WARNING ...\\n\")\n    if interpolation is None: interpolation=DEFAULT_INTERPOLATION\n        \n    # Open the image with PIL\n    tmp_img = Image.open(img_path)\n    \n    # Get original dimensions and calculate new dimensions\n    original_dimensions = tmp_img.size\n    downsample_dimensions = (int(original_dimensions[0]/downsample_by), int(original_dimensions[1]/downsample_by))\n    \n    # Resize the image\n    tmp_img.thumbnail(size=downsample_dimensions, \n                      resample=interpolation,\n                      reducing_gap=reducing_gap)\n    \n    # Return either np.ndarray or PIL.Image object\n    return np.asarray(tmp_img) if as_numpy else tmp_img","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_slide_tiles(img_path, tile_size=(384,384), stride=(256, 256), downsample_by=4, drop_empty=True, drop_near_empty=True, near_empty_sum_thresh=2500, near_empty_bg_thresh=170, return_downsampled_image=False):\n    \"\"\" Tile a WSI\n    \n    Args:\n        img_path (str): Path to .tif file to be downsampled \n        tile_size (tuple of ints, optional): The dimensions of the tile/patches to be extracted\n        stride (tuple of ints, optional): The stride between tile/patches. \n            i.e. a stride of two: xx____, __xx__, ____xx\n        downsample_by (int, optional): How many times smaller should resultant \n            image be. i.e. image_size*(1/downsample_by) = new_size\n        drop_empty (bool, optional): Whether or not to drop images that are determined to be empty\n            i.e. ALL the pixels in the tile are determined to be background pixels\n        drop_near_empty (bool, optional): Whether or not to drop images that are determined to be \"mostly\" empty\n            i.e. MOST of the pixels in the tile are determined to be background pixels\n        near_empty_sum_thresh (int, optional): The threshold to use to determine what \"MOST\"\n            means in the context of background vs forgeround.\n            i.e. How many pixels are allowed to be background pixels before we drop?\n        near_empty_bg_thresh (int, optional): What qualifies as the threshold at which if the pixel\n            is brighter than this it is considered background... wheras lower than this is foreground\n            (I AM AWARE FOREGROUND WILL YIELD SOME PERCENTAGE OF \"BACKGROUND\" PIXELS)\n        return_downsampled_image (bool, optional): Whether we return the downsampled numpy array\n    \n    Returns:\n        A list of tiles – np.ndarray of shape `tile_size`\n    \"\"\"\n    \n    o_img = pil_open_downscaled_slide(img_path, downsample_by)\n    (orig_h, orig_w), (tile_h, tile_w), (stride_w, stride_h) = o_img.shape[:-1], tile_size, stride\n    \n    tile_map, non_bg_sum = {}, near_empty_sum_thresh\n    \n    row_steps = range(0, orig_h+stride_h, stride_h)\n    col_steps = range(0, orig_w+stride_w, stride_w)\n    n_steps = len(row_steps)*len(col_steps)\n    for j, y in enumerate(row_steps):\n        for i, x in enumerate(col_steps):\n            tile = o_img[y:(y+tile_h), x:(x+tile_w)]\n            if drop_empty or drop_near_empty: \n                non_bg_sum = np.where(tile>near_empty_bg_thresh, 0, 1).sum()\n            # Condition to display XXX \n            if (drop_empty and non_bg_sum==0) or (drop_near_empty and non_bg_sum<near_empty_sum_thresh): \n                if n_steps<=999: print(\" XXX \", end=\"\")\n            else: # Else to display the tile number\n                loc_str = f\" {j*len(col_steps)+i+1:>03} \"\n                tile_map[loc_str]=tile\n                if n_steps<=999: print(loc_str, end=\"\")\n        if n_steps<=999: print(\"\\n\")\n        \n    if return_downsampled_image:\n        return tile_map, o_img\n    else:\n        return tile_map","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_and_plot(pd_row, draw_grid=True, grid_stride=(256,256), grid_size=(384,384), draw_text=True):\n    def _add_text(img, annot, color, pt1, pt2, **kwargs):\n            text_width, text_height = cv2.getTextSize(annot, FONT, FONT_SCALE, FONT_THICKNESS)[0]\n            annot_loc = (int(round(pt1[0]+((pt2[0]-pt1[0])/2)-(text_width/2))), int(round(pt1[1]+((pt2[1]-pt1[1])/2)+(text_height/2))))\n            img = cv2.putText(img, annot, annot_loc, FONT, FONT_SCALE, color, FONT_THICKNESS, FONT_LINE_TYPE)\n            return img\n        \n    print(f\"\\n... PROCESSING IMAGE {pd_row.image_id.values[0]} - {pd_row.image_size.values[0]} ...\\n\")\n    t1 = time.time()\n    image_tile_map, downsampled_image = get_slide_tiles(pd_row.image_path.values[0], return_downsampled_image=True)\n    \n    n_tiles = len(image_tile_map)\n    print(f\"\\nIMAGE PROCESSING ELAPSED TIME: {time.time()-t1:.4f} SECONDS\")\n    print(\"RESULTANT NUMBER OF TILES: \", n_tiles)\n    \n    print(\"\\n\\n... ORIGINAL IMAGE - DOWNSAMPLED BY A FACTOR OF 4 ...\\n\\n\")\n    w_to_h_ratio = pd_row.image_width.values[0]/pd_row.image_height.values[0]\n    plt.figure(figsize=(int(20*w_to_h_ratio), 20) if w_to_h_ratio>1 else (20, int(20/w_to_h_ratio)))\n    plt.axis(False)\n    plt.title(f\"Complete Downsampled Image – {pd_row.image_id.values[0]}\", fontweight=\"bold\")\n    if draw_grid:\n        clr_cnt, c_map, clrs, draw_map_fg, draw_map_bg = 0, plt.get_cmap(\"rainbow\"), [], [], []\n        row_steps = range(0, downsampled_image.shape[0]+grid_stride[1], grid_stride[1])\n        col_steps = range(0, downsampled_image.shape[1]+grid_stride[0], grid_stride[0])\n        clrs = [[int(x*255) for x in c_map((i+1)/len(image_tile_map.keys()))] for i in range(len(image_tile_map.keys()))]\n        draw_image = downsampled_image.copy()\n        if draw_text: \n            draw_annots, FONT, FONT_SCALE, FONT_THICKNESS, FONT_LINE_TYPE = [], cv2.FONT_HERSHEY_SIMPLEX, 1.56789, 3, cv2.LINE_AA\n            \n        for j, y in enumerate(row_steps):\n            for i, x in enumerate(col_steps):\n                if f\" {j*len(col_steps)+i+1:>03} \" in image_tile_map.keys():\n                    draw_map_fg.append(dict(pt1=(x,y), pt2=(x+grid_size[0], y+grid_size[1]), color=clrs[clr_cnt], thickness=3, annot=f\" {j*len(col_steps)+i+1:>03} \"))\n                    clr_cnt +=1\n                else:\n                    draw_map_bg.append(dict(pt1=(x,y), pt2=(x+grid_size[0], y+grid_size[1]), color=(10,10,10), thickness=2, annot=f\" {j*len(col_steps)+i+1:>03} \"))                \n                    \n        for draw_map in draw_map_bg+draw_map_fg:\n            draw_image = cv2.rectangle(draw_image, draw_map[\"pt1\"], draw_map[\"pt2\"], draw_map[\"color\"], draw_map[\"thickness\"])\n            if draw_text:\n                _add_text(draw_image, draw_map[\"annot\"], draw_map[\"color\"], draw_map[\"pt1\"], draw_map[\"pt2\"])\n        ### STRIDE LINES ###\n        #   hline_list = list(range(0, downsampled_image.shape[0]+sample_grid_stride[0], sample_grid_stride[0]))\n        #   vline_list = list(range(0, downsampled_image.shape[1]+sample_grid_stride[1], sample_grid_stride[1]))\n        #   plt.hlines(hline_list, xmin=0, xmax=sample_grid_stride[0]*(len(vline_list)-1), colors=\"k\")\n        #   plt.vlines(vline_list, ymin=0, ymax=sample_grid_stride[1]*(len(vline_list)-1), colors=\"k\")\n    \n    plt.imshow(downsampled_image if not draw_grid else draw_image)\n    plt.tight_layout()\n    plt.show()\n\n    print(f\"\\n\\n\\n... PLOTTING FIRST {min(16, n_tiles)} OF {n_tiles} TILES FROM IMAGE ...\\n\")\n    plt.figure(figsize=(20,20))\n    for i, (loc_str, tile) in enumerate(image_tile_map.items()):\n        if i==16: break\n        plt.subplot(4,4,i+1)\n        plt.title(f\"Tile Location: {loc_str}\", fontweight=\"bold\")\n        plt.axis(False)\n        plt.imshow(tile)\n    plt.tight_layout()\n    plt.show()","metadata":{},"execution_count":null,"outputs":[]}]}