{"cells":[{"metadata":{},"cell_type":"markdown","source":"# **About Notebook**\n\nThis notebook explores the data through the lens of Whole Slide Images - i.e. across sizes, total pixels and a couple of other provided data columns in the train csv file using plotly.\n\nLet me know if you found these valuable.\n","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"!pip install itk --quiet\n!pip install itkwidgets --quiet","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np \nimport pandas as pd \n\nimport openslide\nimport gc\nimport matplotlib.pyplot as plt\nfrom collections import defaultdict\n\nfrom PIL import Image\nfrom tqdm import tqdm\n\nimport plotly\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\n\n\nfrom IPython.display import FileLinks\n\nimport itk\nimport itkwidgets\nfrom ipywidgets import interact , interactive , IntSlider , ToggleButtons\nfrom ipywidgets import interact\nplotly.offline.init_notebook_mode (True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Path to the Directory**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images_dir = '../input/prostate-cancer-grade-assessment/train_images/'\ntrain_images     = os.listdir (train_images_dir)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv = pd.read_csv ('../input/prostate-cancer-grade-assessment/train.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Dataframe construction with Image Dimensions**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images  = []\n\nfor i , image in enumerate(tqdm(os.listdir (train_images_dir))):\n    img             =  image \n    img_size_MB     = f\"{os.stat(train_images_dir+image).st_size / 1024 **2 : 1.2f} \" \n    wsi             = openslide.OpenSlide (train_images_dir + image)\n    train_images.append((img ,  \n                         img_size_MB ,\n                         wsi.level_dimensions[0] , \n                         wsi.level_dimensions[1] , \n                         wsi.level_dimensions[2] , \n                         np.product(wsi.level_dimensions[0]) * 3, \n                         np.product(wsi.level_dimensions[1]) * 3, \n                         np.product(wsi.level_dimensions[2]) * 3\n                        ))\n    gc.collect()\n    \n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Converting the above results to a DataFrame**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images = pd.DataFrame ( train_images , \n                              columns = ['img' , \n                                  'Image Size (MB)' ,\n                                  'Image Shape Level0' ,\n                                  'Image Shape Level1' , \n                                  'Image Shape Level2' , \n                                  'Total Pixels Level0' , \n                                  'Total Pixels Level1' , \n                                  'Total Pixels Level2']\n                            )","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**The below dataframe details the information across WSI's in the training dataset.**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images['Image Size (MB)'] = train_images['Image Size (MB)'].astype('float32')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Remove .tiff from the 'img' column in Dataframe**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def image_name(img) :\n    return img.split('.')[0]\n\ntrain_images['img'] = train_images['img'].map(image_name)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_images.shape  , train_csv.shape , train_images['img'].nunique()  , train_csv['image_id'].nunique()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Merging the Whole slide images information with the Train DataFrame.**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = train_csv.merge (train_images , \n                            left_on = \"image_id\" , \n                            right_on = \"img\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Drop duplicate column\ntrain_df.drop('img', axis = 1, inplace= True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Above dataframe will be used futher for data insights through visualization","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"**Dropping the intermediate created datasets**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"del train_images , train_csv","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Image Size Distribution","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"**Fetching the image width and height for plots**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_width (image) :\n    return image[0]\ndef get_height (image) :\n    return image[1]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Below are the scatter plots of Image pixels across height and width for all 3 levels**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"width  = train_df['Image Shape Level0'].map(get_width)\nheight = train_df['Image Shape Level0'].map(get_height) \n\nfig    = px.scatter (train_df , \n            x =  width , \n            y = height , \n            color = width,\n           title = 'Image Size in Pixels - Level 0 (highest) Resolution)')\n\nfig.update_layout ( yaxis=dict(title_text=\"Height\") , \n                    xaxis=dict(title_text=\"Width\") , \n                    title_font_family=\"Open Sans\"\n                  )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"width  = train_df['Image Shape Level1'].map(get_width)\nheight = train_df['Image Shape Level1'].map(get_height) \nfig    =  px.scatter (train_df , \n            x =  width , \n            y = height , \n            color = width , \n            title = 'Image Size in Pixels - Level 1 Resolution)')\n\nfig.update_layout (yaxis=dict(title_text=\"Height\") , \n                    xaxis=dict(title_text=\"Width\") , \n                    title_font_family=\"Open Sans\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"width  = train_df['Image Shape Level2'].map(get_width)\nheight = train_df['Image Shape Level2'].map(get_height) \nfig    = px.scatter (train_df , \n            x =  width , \n            y = height , \n            color = width , \n            title = 'Image Size in Pixels - Level 2 (lowest) Resolution)')\nfig.update_layout (yaxis=dict(title_text=\"Height\") , \n                    xaxis=dict(title_text=\"Width\") , \n                    title_font_family=\"Open Sans\")\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**What do the Image Size plots tell ?**\n1. Image width & height for level 0 lies in range of tens of thousands , level1 until tens of 1000's and the last one i.e. level2 is concentrated in range of a few thousands majorily.\n2. There are a few outliers in terms of sizes among above 3 levels.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# **Toggle Level**","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"**Below one is an interactive plot to switch among 3 levels quickly and get an overview of the distribution**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def change_level (level):\n    width  = train_df[f'Image Shape Level{level}'].map(get_width)\n    height = train_df[f'Image Shape Level{level}'].map(get_height) \n    fig = px.scatter (train_df , \n            x =  width , \n            y = height , \n            color = width , \n            title = f'Image Size in Pixels - Level {level} (Lowest Resolution)')\n    \n    fig.update_layout ( yaxis=dict(title_text=\"Height\") , \n                    xaxis=dict(title_text=\"Width\") , \n                    title_font_family=\"Open Sans\")\n    fig.show()\n    return level","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"interact (change_level , level = (0 , 2))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# **Pixel Distribution**\n","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"**Below plots visualise the pixel**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram (train_df ,\n              x = ['Total Pixels Level0'] ,\n              hover_data = ['gleason_score' , 'isup_grade'],\n              color = 'data_provider' ,\n              marginal = 'rug',\n              title = 'Pixel Distribution at Level 0')\n\nfig.update_layout ( yaxis=dict(title_text=\"Images Count\") , \n                    xaxis=dict(title_text=\"Total Number of Pixels\") , \n                    title_font_family=\"Open Sans\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram (train_df ,\n              x = ['Total Pixels Level1'] ,\n              hover_data = ['gleason_score' , 'isup_grade'],\n              color = 'data_provider' ,\n              marginal = 'rug',\n              title = 'Pixel Distribution at Level 1')\nfig.update_layout ( yaxis=dict(title_text=\"Images Count\") , \n                    xaxis=dict(title_text=\"Pixel Size\") , \n                    title_font_family=\"Open Sans\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.histogram (train_df ,\n              x = ['Total Pixels Level2'] ,\n              hover_data = ['gleason_score' , 'isup_grade'],\n              color = 'data_provider' ,\n              marginal = 'rug',\n              title = 'Pixel Distribution at Level 2'\n                   )\n\nfig.update_layout ( yaxis=dict(title_text=\"Images Count\") , \n                    xaxis=dict(title_text=\"Pixel Size\") , \n                    title_font_family=\"Open Sans\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**What do the Total Pixel's plots tell us ?**\n1. Image data size do vary as provided by the 2 data providers.\n2. In all the 3 levels , Karolinska provided datasets are comparatively larger than in size.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"**Above distinction among data providers can be more clearly seen in the below scatter plot**","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# **Toggle Level**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def change_level (level):\n    width  = train_df[f'Image Shape Level{level}'].map(get_width)\n    height = train_df[f'Image Shape Level{level}'].map(get_height) \n    fig = px.scatter (train_df , \n            x =  width , \n            y = height , \n            color = train_df['data_provider'] , \n            title = f'Image Size Distribution across Data providers')\n    fig.update_traces(marker=dict(size=12,\n                      line=dict(width=2,color='DarkSlateGrey')),\n                      selector=dict(mode='markers')\n                  )\n\n    fig.show()\n    return level","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"interact (change_level , level = (0 , 2))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Note : Above interactive plot might not be visible in comit notebook.**","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"**Below plot analyses if the Gradings vary with image size (i.e. Total number of Pixels) at a given resolution level ?**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = make_subplots(rows=1, \n                    cols=3 , \n                    subplot_titles=(\"Level 0\", \n                                    \"Level 1\" , \n                                    \"Level 2\")\n                   )\nfig.add_trace(go.Violin(\n                        x = train_df['isup_grade'] ,\n                        y = train_df['Total Pixels Level0'], \n                        points = 'all', name = \"Level 0\"\n                ), row = 1, col = 1 )\n\nfig.add_trace(go.Violin(\n                        x = train_df['isup_grade'] ,\n                        y = train_df['Total Pixels Level1'], \n                        points = 'all' , name = \"Level 1\"\n                ) , row = 1, col = 2 )\n\n\nfig.add_trace(go.Violin(\n                        x = train_df['isup_grade'] ,\n                        y = train_df['Total Pixels Level2'], \n                        points = 'all', name = \"Level 2\" \n                ), row =1  , col = 3 )\n\nfig.update_layout(\n    autosize=True , \n    width=2000,\n    height=500 , \n    title = 'Total Pixels distribution by Grading '\n)\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Here, we note that distribution is similar across the image sizes.","execution_count":null}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}