{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-05T12:32:52.020785Z","iopub.execute_input":"2023-04-05T12:32:52.021412Z","iopub.status.idle":"2023-04-05T12:32:52.359947Z","shell.execute_reply.started":"2023-04-05T12:32:52.021364Z","shell.execute_reply":"2023-04-05T12:32:52.358877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport cv2\nimport keras.backend as K \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom pprint import pprint\nfrom collections import defaultdict\nimport openslide\nfrom openslide import OpenSlide\nfrom glob import glob\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Dropout, Flatten, Dense\nfrom tensorflow.keras.layers import GlobalMaxPooling2D\nfrom keras.models import load_model","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:32:52.362448Z","iopub.execute_input":"2023-04-05T12:32:52.363233Z","iopub.status.idle":"2023-04-05T12:33:01.341311Z","shell.execute_reply.started":"2023-04-05T12:32:52.363189Z","shell.execute_reply":"2023-04-05T12:33:01.340186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/mayo-clinic-strip-ai/train.csv\")\ntest_df = pd.read_csv(\"../input/mayo-clinic-strip-ai/test.csv\")\nother_df = pd.read_csv(\"../input/mayo-clinic-strip-ai/other.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.342999Z","iopub.execute_input":"2023-04-05T12:33:01.343856Z","iopub.status.idle":"2023-04-05T12:33:01.368939Z","shell.execute_reply.started":"2023-04-05T12:33:01.343805Z","shell.execute_reply":"2023-04-05T12:33:01.367933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.37215Z","iopub.execute_input":"2023-04-05T12:33:01.37256Z","iopub.status.idle":"2023-04-05T12:33:01.38314Z","shell.execute_reply.started":"2023-04-05T12:33:01.372519Z","shell.execute_reply":"2023-04-05T12:33:01.382077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.385204Z","iopub.execute_input":"2023-04-05T12:33:01.385749Z","iopub.status.idle":"2023-04-05T12:33:01.394153Z","shell.execute_reply.started":"2023-04-05T12:33:01.385697Z","shell.execute_reply":"2023-04-05T12:33:01.393002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.395919Z","iopub.execute_input":"2023-04-05T12:33:01.396451Z","iopub.status.idle":"2023-04-05T12:33:01.413709Z","shell.execute_reply.started":"2023-04-05T12:33:01.396411Z","shell.execute_reply":"2023-04-05T12:33:01.412658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.415338Z","iopub.execute_input":"2023-04-05T12:33:01.415708Z","iopub.status.idle":"2023-04-05T12:33:01.437292Z","shell.execute_reply.started":"2023-04-05T12:33:01.41567Z","shell.execute_reply":"2023-04-05T12:33:01.436026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Memory saving function credit to https://www.kaggle.com/gemartin/load-data-reduce-memory-usage\n# Avoid memory intensive tasks which could cause the kernel to restart\n\ndef reduce_mem_usage(df): \n    for col in df.columns:\n        col_type = df[col].dtype\n\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    return df.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.439059Z","iopub.execute_input":"2023-04-05T12:33:01.439483Z","iopub.status.idle":"2023-04-05T12:33:01.453405Z","shell.execute_reply.started":"2023-04-05T12:33:01.439432Z","shell.execute_reply":"2023-04-05T12:33:01.452088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nprint('Train_df')\nprint(reduce_mem_usage(train_df))\nprint('\\n')\nprint('Test_df')\nprint(reduce_mem_usage(test_df))\nprint('\\n')\nprint('Other_df')\nprint(reduce_mem_usage(other_df))\nprint('\\n')\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.455293Z","iopub.execute_input":"2023-04-05T12:33:01.455838Z","iopub.status.idle":"2023-04-05T12:33:01.481902Z","shell.execute_reply.started":"2023-04-05T12:33:01.4558Z","shell.execute_reply":"2023-04-05T12:33:01.480729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.shape)\nprint(test_df.shape)\nprint(other_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.48682Z","iopub.execute_input":"2023-04-05T12:33:01.487125Z","iopub.status.idle":"2023-04-05T12:33:01.496864Z","shell.execute_reply.started":"2023-04-05T12:33:01.487096Z","shell.execute_reply":"2023-04-05T12:33:01.495497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['image_path'] = train_df['image_id'].apply(lambda x: \"../input/mayo-clinic-strip-ai/train/\" + x + \".tif\")\ntrain_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.498946Z","iopub.execute_input":"2023-04-05T12:33:01.499344Z","iopub.status.idle":"2023-04-05T12:33:01.515668Z","shell.execute_reply.started":"2023-04-05T12:33:01.499304Z","shell.execute_reply":"2023-04-05T12:33:01.514143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.517504Z","iopub.execute_input":"2023-04-05T12:33:01.518811Z","iopub.status.idle":"2023-04-05T12:33:01.526919Z","shell.execute_reply.started":"2023-04-05T12:33:01.51877Z","shell.execute_reply":"2023-04-05T12:33:01.525496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum() ","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.52916Z","iopub.execute_input":"2023-04-05T12:33:01.530067Z","iopub.status.idle":"2023-04-05T12:33:01.540721Z","shell.execute_reply.started":"2023-04-05T12:33:01.530025Z","shell.execute_reply":"2023-04-05T12:33:01.539165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['image_id'].nunique() #checking for no of unique","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.54284Z","iopub.execute_input":"2023-04-05T12:33:01.543695Z","iopub.status.idle":"2023-04-05T12:33:01.551972Z","shell.execute_reply.started":"2023-04-05T12:33:01.543613Z","shell.execute_reply":"2023-04-05T12:33:01.550533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['image_id'].value_counts().sum()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.554048Z","iopub.execute_input":"2023-04-05T12:33:01.554842Z","iopub.status.idle":"2023-04-05T12:33:01.564119Z","shell.execute_reply.started":"2023-04-05T12:33:01.554787Z","shell.execute_reply":"2023-04-05T12:33:01.562721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## No of unique values in each column\nprint('Unique Values for: ')\ntrain_df.nunique()\n\nprint('\\n')\n\nnu = train_df.nunique().reset_index()\nnu.columns = ['feature','nunique']\nplt.figure(figsize=(12,4))\nax = sns.barplot(x='feature', y='nunique', data=nu)\nax.bar_label(ax.containers[0])\nax.set_title('No of unique values in each column')","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.566123Z","iopub.execute_input":"2023-04-05T12:33:01.567011Z","iopub.status.idle":"2023-04-05T12:33:01.870048Z","shell.execute_reply.started":"2023-04-05T12:33:01.566946Z","shell.execute_reply":"2023-04-05T12:33:01.869019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['center_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.871594Z","iopub.execute_input":"2023-04-05T12:33:01.872247Z","iopub.status.idle":"2023-04-05T12:33:01.882821Z","shell.execute_reply.started":"2023-04-05T12:33:01.872206Z","shell.execute_reply":"2023-04-05T12:33:01.88165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nsns.countplot(data=train_df, x='center_id', palette='flare', order=train_df['center_id'].value_counts(ascending=True).index)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:01.884531Z","iopub.execute_input":"2023-04-05T12:33:01.885265Z","iopub.status.idle":"2023-04-05T12:33:02.145396Z","shell.execute_reply.started":"2023-04-05T12:33:01.885217Z","shell.execute_reply":"2023-04-05T12:33:02.144335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Take note of the min and max values for number of slides per patient","metadata":{}},{"cell_type":"code","source":"train_df['patient_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:02.147012Z","iopub.execute_input":"2023-04-05T12:33:02.147653Z","iopub.status.idle":"2023-04-05T12:33:02.158415Z","shell.execute_reply.started":"2023-04-05T12:33:02.147612Z","shell.execute_reply":"2023-04-05T12:33:02.15694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['patient_id'].value_counts().describe()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:02.160224Z","iopub.execute_input":"2023-04-05T12:33:02.160797Z","iopub.status.idle":"2023-04-05T12:33:02.173179Z","shell.execute_reply.started":"2023-04-05T12:33:02.160759Z","shell.execute_reply":"2023-04-05T12:33:02.172031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[train_df['patient_id']=='3d10be']","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:02.174983Z","iopub.execute_input":"2023-04-05T12:33:02.175379Z","iopub.status.idle":"2023-04-05T12:33:02.19179Z","shell.execute_reply.started":"2023-04-05T12:33:02.175337Z","shell.execute_reply":"2023-04-05T12:33:02.190863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,6))\nsns.countplot(data=train_df, x='image_num', order=train_df['image_num'].value_counts().index, palette='mako')","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:02.192914Z","iopub.execute_input":"2023-04-05T12:33:02.193337Z","iopub.status.idle":"2023-04-05T12:33:02.421731Z","shell.execute_reply.started":"2023-04-05T12:33:02.193302Z","shell.execute_reply":"2023-04-05T12:33:02.420745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" **Most patients have only 1 image**","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d=train_df['label'].value_counts().reset_index()\nd['percentage']=d['label'].apply(lambda x: (x/sum(train_df['label'].value_counts()))*100)\nd","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:02.42338Z","iopub.execute_input":"2023-04-05T12:33:02.423767Z","iopub.status.idle":"2023-04-05T12:33:02.438688Z","shell.execute_reply.started":"2023-04-05T12:33:02.423712Z","shell.execute_reply":"2023-04-05T12:33:02.437486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d=d.rename({'index': 'label', 'label': 'count'}, axis='columns')\nd","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:02.440585Z","iopub.execute_input":"2023-04-05T12:33:02.441394Z","iopub.status.idle":"2023-04-05T12:33:02.453408Z","shell.execute_reply.started":"2023-04-05T12:33:02.441352Z","shell.execute_reply":"2023-04-05T12:33:02.452191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_df, x='label')","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:02.456149Z","iopub.execute_input":"2023-04-05T12:33:02.456918Z","iopub.status.idle":"2023-04-05T12:33:02.637796Z","shell.execute_reply.started":"2023-04-05T12:33:02.456873Z","shell.execute_reply":"2023-04-05T12:33:02.63683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Unbalanced Dataset, CE makes up ~73% of the entire dataset\nWe can use data augmentation on the image dataset for deep learning.","metadata":{}},{"cell_type":"markdown","source":"**Reading a sample TIF image**","metadata":{}},{"cell_type":"code","source":"from PIL import Image","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:02.63938Z","iopub.execute_input":"2023-04-05T12:33:02.639717Z","iopub.status.idle":"2023-04-05T12:33:02.644915Z","shell.execute_reply.started":"2023-04-05T12:33:02.639682Z","shell.execute_reply":"2023-04-05T12:33:02.643904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nImage.MAX_IMAGE_PIXELS = None\nsize = (512,512)\n\n\nsample_slides = train_df.sample(n=3, random_state=1) \nfig, axs = plt.subplots(1, 3, figsize = (12, 12))\n\nfor i in range(len(sample_slides.index)): \n    \n    slide = Image.open(sample_slides.iloc[i]['image_path'])\n    slide.thumbnail(size)\n    axs[i].set_title(f'Image Size: {slide.size}')\n    axs[i].imshow(slide)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:02.646568Z","iopub.execute_input":"2023-04-05T12:33:02.647305Z","iopub.status.idle":"2023-04-05T12:33:49.067972Z","shell.execute_reply.started":"2023-04-05T12:33:02.647268Z","shell.execute_reply":"2023-04-05T12:33:49.06693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images = glob(\"/kaggle/input/mayo-clinic-strip-ai/train/*\")\ntest_images = glob(\"/kaggle/input/mayo-clinic-strip-ai/test/*\")\nother_images = glob(\"/kaggle/input/mayo-clinic-strip-ai/other/*\")\nprint(f\"Number of images in a training set: {len(train_images)}\")\nprint(f\"Number of images in a training set: {len(test_images)}\")\nprint(f\"Number of other: {len(other_images)}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:49.076067Z","iopub.execute_input":"2023-04-05T12:33:49.076724Z","iopub.status.idle":"2023-04-05T12:33:49.090903Z","shell.execute_reply.started":"2023-04-05T12:33:49.076684Z","shell.execute_reply":"2023-04-05T12:33:49.089995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_prop = defaultdict(list)\nfor i, path in enumerate(train_images):\n    img_path = train_images[i]\n    slide = OpenSlide(img_path)    \n    img_prop['image_id'].append(img_path[-12:-4])\n    img_prop['width'].append(slide.dimensions[0])\n    img_prop['height'].append(slide.dimensions[1])\n    img_prop['size'].append(round(os.path.getsize(img_path) / 1e6, 2))\n    img_prop['path'].append(img_path)\nimage_data = pd.DataFrame(img_prop)\nimage_data['img_aspect_ratio'] = image_data['width']/image_data['height']\nimage_data.sort_values(by='image_id', inplace=True)\nimage_data.reset_index(inplace=True, drop=True)\nimage_data = image_data.merge(train_df, on='image_id')\nimage_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:33:49.092431Z","iopub.execute_input":"2023-04-05T12:33:49.092788Z","iopub.status.idle":"2023-04-05T12:34:05.234303Z","shell.execute_reply.started":"2023-04-05T12:33:49.092751Z","shell.execute_reply":"2023-04-05T12:34:05.233151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfig, ax = plt.subplots(1,2, figsize=(16,5))\nsns.histplot(x='size', data = image_data, bins=100, ax=ax[0])\nax[0].set_title(\"Distribution of size\"), ax[0].set_ylabel(\"%\")\nsns.histplot(x='img_aspect_ratio', data = image_data, bins=100, ax=ax[1])\nax[1].set_title(\"Image aspect ratio\"), ax[1].set_ylabel(\"%\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:34:05.235738Z","iopub.execute_input":"2023-04-05T12:34:05.236815Z","iopub.status.idle":"2023-04-05T12:34:05.874081Z","shell.execute_reply.started":"2023-04-05T12:34:05.23677Z","shell.execute_reply":"2023-04-05T12:34:05.872934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reading_tiff(image):\n    Reading_Image = cv2.cvtColor(cv2.imread(image),cv2.COLOR_BGR2RGB)\n    return Reading_Image","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:34:05.876024Z","iopub.execute_input":"2023-04-05T12:34:05.876689Z","iopub.status.idle":"2023-04-05T12:34:05.882276Z","shell.execute_reply.started":"2023-04-05T12:34:05.876646Z","shell.execute_reply":"2023-04-05T12:34:05.880781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Reading_Image=reading_tiff('../input/mayo-clinic-strip-ai/train/31adaa_0.tif')\nfigure = plt.figure(figsize=(20,20))\nplt.imshow(Reading_Image)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:34:05.884388Z","iopub.execute_input":"2023-04-05T12:34:05.885203Z","iopub.status.idle":"2023-04-05T12:34:58.211521Z","shell.execute_reply.started":"2023-04-05T12:34:05.885158Z","shell.execute_reply":"2023-04-05T12:34:58.207788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"slide = OpenSlide('/kaggle/input/mayo-clinic-strip-ai/train/026c97_0.tif') # opening a full slide\n\nregion = (2500, 2000) # location of the top left pixel\nlevel = 0 # level of the picture (we have only 0)\nsize = (3500, 3500) # region size in pixels\n\nregion = slide.read_region(region, level, size)\nimage = region.resize((512, 512))\nplt.figure(figsize=(10, 10))\nplt.imshow(image)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:34:58.215607Z","iopub.execute_input":"2023-04-05T12:34:58.219213Z","iopub.status.idle":"2023-04-05T12:35:01.139644Z","shell.execute_reply.started":"2023-04-05T12:34:58.219157Z","shell.execute_reply":"2023-04-05T12:35:01.137034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\ncontinuous_color_scheme = px.colors.sequential.Sunset\ndiscrete_color_scheme = px.colors.qualitative.Pastel1\n# opacity\nplots_opacity = 0.8","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:35:01.14111Z","iopub.execute_input":"2023-04-05T12:35:01.142325Z","iopub.status.idle":"2023-04-05T12:35:03.34886Z","shell.execute_reply.started":"2023-04-05T12:35:01.142268Z","shell.execute_reply":"2023-04-05T12:35:03.347821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n# Q: Which center is the most polular one?\n# A: 11\n# some centers have only 20-30 images taken\n# all centers have data of both classes\n\n# distribution over centers\ndistribution_over_classes = train_df['center_id'].value_counts().reset_index().rename(columns={'index':'Center', 'center_id':'Count'})\nfig = px.bar(distribution_over_classes, x=\"Center\", y=\"Count\", title='Comparison of number of examples in test dataset by Center'\n             , text='Count', color = 'Center', \n             color_continuous_scale= continuous_color_scheme\n             )  \n             #color_discrete_sequence = generate_color_sequence(2) )#cmocean.cm.thermal)\nfig.update_traces(texttemplate='%{text:.0s}', textposition='outside') # prints values above bars\nfig.show()\n\n# distribution over centers and classes \nfig = px.bar(train_df.groupby(by=['center_id', 'label'], as_index=False).count().rename(columns={'image_id':'Count'}), x=\"center_id\", y=\"Count\", \n             color=\"label\", \n             title=\"Data distribution over centers and labels\",  \n             color_discrete_sequence=discrete_color_scheme,\n             opacity=plots_opacity\n             )\nfig.show()\n\n# the same data can be shown as sunburst plot\nfig = px.sunburst(train_df[['center_id', 'label', 'image_id']].groupby(by=['center_id', 'label'], as_index=False).count().rename(columns={'image_id':'Count'}),\n                  path=['center_id', 'label'], values='Count', color_discrete_sequence=discrete_color_scheme)\nfig.show()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:35:03.350474Z","iopub.execute_input":"2023-04-05T12:35:03.350828Z","iopub.status.idle":"2023-04-05T12:35:04.699732Z","shell.execute_reply.started":"2023-04-05T12:35:03.350789Z","shell.execute_reply":"2023-04-05T12:35:04.698601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Q: do patients have the same label result after reexaminations?\n# A: Yes\n# patients with more that 1 image\npatients_with_several_images_grouped = train_df[['patient_id', 'image_num']].groupby(by=['patient_id'], as_index=False).count().query('image_num > 1')\n# patients with more that 1 image grouped by patient_id, label\npatients_with_several_images = train_df[train_df['patient_id'].isin(patients_with_several_images_grouped['patient_id'])][['patient_id', 'label', 'image_id']].groupby(by=['patient_id', 'label'], as_index=False).count()\n# sql-like join\npatients_with_several_images = pd.merge(left=patients_with_several_images, right=patients_with_several_images_grouped, how=\"inner\", on = 'patient_id')\n# if image_id <> image number that the same patient has images with different labels\npatients_with_several_images.query('image_id != image_num')# there are no such cases in dataset","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:35:04.701276Z","iopub.execute_input":"2023-04-05T12:35:04.701834Z","iopub.status.idle":"2023-04-05T12:35:04.738211Z","shell.execute_reply.started":"2023-04-05T12:35:04.701792Z","shell.execute_reply":"2023-04-05T12:35:04.737282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Q: Are the reexaminations taken in the same center?\n# A: Yes\n# patients with more that 1 image\npatients_with_several_images_grouped = train_df[['patient_id', 'image_num']].groupby(by=['patient_id'], as_index=False).count().query('image_num > 1') \n# patients with more that 1 image grouped by patient_id, center_id\npatients_with_several_images = train_df[train_df['patient_id'].isin(patients_with_several_images_grouped['patient_id'])][['patient_id', 'center_id', 'image_id']].groupby(by=['patient_id', 'center_id'], as_index=False).count()\n# sql-like join\npatients_with_several_images = pd.merge(left=patients_with_several_images, right=patients_with_several_images_grouped, how=\"inner\", on = 'patient_id')\n# if image_id <> image number that the same patient has images with different labels\npatients_with_several_images.query('image_id != image_num')# there are no such cases in dataset","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:35:04.739652Z","iopub.execute_input":"2023-04-05T12:35:04.740012Z","iopub.status.idle":"2023-04-05T12:35:04.767645Z","shell.execute_reply.started":"2023-04-05T12:35:04.739974Z","shell.execute_reply":"2023-04-05T12:35:04.766463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[[\"label\"]].value_counts().plot.pie(autopct='%1.1f%%',ylabel=\"label\")","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:35:04.769349Z","iopub.execute_input":"2023-04-05T12:35:04.769748Z","iopub.status.idle":"2023-04-05T12:35:04.889476Z","shell.execute_reply.started":"2023-04-05T12:35:04.769707Z","shell.execute_reply":"2023-04-05T12:35:04.888221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Unique patients having CE condition', train_df.loc[train_df['label']=='CE']['patient_id'].nunique() , 'of total',train_df['patient_id'].nunique() )\nprint('Unique patients having LAA condition', train_df.loc[train_df['label']=='LAA']['patient_id'].nunique(),'of total',train_df['patient_id'].nunique() )","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:35:04.890974Z","iopub.execute_input":"2023-04-05T12:35:04.891312Z","iopub.status.idle":"2023-04-05T12:35:04.906173Z","shell.execute_reply.started":"2023-04-05T12:35:04.891275Z","shell.execute_reply":"2023-04-05T12:35:04.904214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.groupby(['patient_id']).agg(distinct_condition=('label','nunique')).reset_index()['distinct_condition'].value_counts() \n\n# None of the patients presented with both conditions","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:35:04.90806Z","iopub.execute_input":"2023-04-05T12:35:04.908455Z","iopub.status.idle":"2023-04-05T12:35:04.931454Z","shell.execute_reply.started":"2023-04-05T12:35:04.908414Z","shell.execute_reply":"2023-04-05T12:35:04.93023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Model training**","metadata":{}},{"cell_type":"code","source":"train_df[\"file_path\"] = train_df[\"image_id\"].apply(lambda x: \"../input/mayo-clinic-strip-ai/train/\" + x + \".tif\")\ntest_df[\"file_path\"]  = test_df[\"image_id\"].apply(lambda x: \"../input/mayo-clinic-strip-ai/test/\" + x + \".tif\")","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:35:04.933198Z","iopub.execute_input":"2023-04-05T12:35:04.933897Z","iopub.status.idle":"2023-04-05T12:35:04.943585Z","shell.execute_reply.started":"2023-04-05T12:35:04.933858Z","shell.execute_reply":"2023-04-05T12:35:04.941773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# labelling CE class as 1 and LAA as 0\ntrain_df[\"target\"] = train_df[\"label\"].apply(lambda x : 1 if x==\"CE\" else 0)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:35:04.945928Z","iopub.execute_input":"2023-04-05T12:35:04.946702Z","iopub.status.idle":"2023-04-05T12:35:04.970351Z","shell.execute_reply.started":"2023-04-05T12:35:04.946662Z","shell.execute_reply":"2023-04-05T12:35:04.96913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Preprocessing¶**","metadata":{}},{"cell_type":"code","source":"%%time\ndef preprocess(image_path):\n    slide=OpenSlide(image_path)\n    region= (2500,2500)    \n    size  = (5000, 5000)\n    image = slide.read_region(region, 0, size)\n    image = image.resize((128, 128))\n    image = np.array(image)    \n    return image\n\nX_train=[]\nfor i in tqdm(train_df['file_path']):\n    x1=preprocess(i)\n    X_train.append(x1)\n\nY_train=[]    \nY_train=train_df['target']","metadata":{"execution":{"iopub.status.busy":"2023-04-05T12:35:04.972061Z","iopub.execute_input":"2023-04-05T12:35:04.972769Z","iopub.status.idle":"2023-04-05T13:22:15.505309Z","shell.execute_reply.started":"2023-04-05T12:35:04.97272Z","shell.execute_reply":"2023-04-05T13:22:15.503482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train=np.array(X_train)\nX_train=X_train/255.0\nY_train = np.array(Y_train)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:22:15.510721Z","iopub.execute_input":"2023-04-05T13:22:15.511056Z","iopub.status.idle":"2023-04-05T13:22:16.14554Z","shell.execute_reply.started":"2023-04-05T13:22:15.511024Z","shell.execute_reply":"2023-04-05T13:22:16.14445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Splitting data\nx_train,x_test,y_train,y_test=train_test_split(X_train,Y_train, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:22:16.147082Z","iopub.execute_input":"2023-04-05T13:22:16.147648Z","iopub.status.idle":"2023-04-05T13:22:16.728188Z","shell.execute_reply.started":"2023-04-05T13:22:16.147598Z","shell.execute_reply":"2023-04-05T13:22:16.726923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Splitting data\nx_train,x_test,y_train,y_test=train_test_split(X_train,Y_train, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:22:16.729736Z","iopub.execute_input":"2023-04-05T13:22:16.730132Z","iopub.status.idle":"2023-04-05T13:22:17.23395Z","shell.execute_reply.started":"2023-04-05T13:22:16.73009Z","shell.execute_reply":"2023-04-05T13:22:17.232866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **CNN approach**","metadata":{}},{"cell_type":"code","source":"def f1_score(y_true, y_pred): \n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n    predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n    precision = true_positives / (predicted_positives + K.epsilon())\n    recall = true_positives / (possible_positives + K.epsilon())\n    f1_val = 2*(precision*recall)/(precision+recall+K.epsilon())\n    return f1_val","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:22:17.235581Z","iopub.execute_input":"2023-04-05T13:22:17.236001Z","iopub.status.idle":"2023-04-05T13:22:17.243386Z","shell.execute_reply.started":"2023-04-05T13:22:17.235934Z","shell.execute_reply":"2023-04-05T13:22:17.242188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras import metrics\n\nmodel = Sequential()\ninput_shape = (128, 128, 4)\n\nmodel.add(Conv2D(filters=64, kernel_size = (3,3), padding = 'valid', activation = 'relu', input_shape = input_shape))\nmodel.add(MaxPooling2D())\nmodel.add(Conv2D(filters=64, kernel_size = (3,3), strides =2, padding = 'valid', activation = 'relu'))\nmodel.add(Conv2D(filters=64, kernel_size = (3,3), strides =2, padding = 'valid', activation = 'relu'))\nmodel.add(Conv2D(filters=64, kernel_size = (3,3), strides =2, padding = 'valid', activation = 'relu'))\nmodel.add(Conv2D(filters=128, kernel_size = (3,3), strides =2, padding = 'valid', activation = 'relu'))\nmodel.add(Conv2D(filters=128, kernel_size = (3,3), strides =2, padding = 'valid', activation = 'relu'))\n\nmodel.add(Dropout(0.13))\nmodel.add(Flatten())\nmodel.add(Dense(256, activation = 'relu'))\nmodel.add(Dropout(0.13))\nmodel.add(Dense(100, activation = 'relu'))\nmodel.add(Dense(50, activation = 'relu'))\nmodel.add(Dense(1 , activation=\"sigmoid\"))\n\nmodel.compile(\n    loss = tf.keras.losses.BinaryCrossentropy(),\n    metrics=[metrics.binary_accuracy,f1_score],\n    optimizer = tf.keras.optimizers.Adam(1e-3))","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:22:17.244907Z","iopub.execute_input":"2023-04-05T13:22:17.245944Z","iopub.status.idle":"2023-04-05T13:22:21.315942Z","shell.execute_reply.started":"2023-04-05T13:22:17.245901Z","shell.execute_reply":"2023-04-05T13:22:21.314899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dot_img_file = 'model.png'\ntf.keras.utils.plot_model(model, to_file=dot_img_file, show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:22:21.317603Z","iopub.execute_input":"2023-04-05T13:22:21.318023Z","iopub.status.idle":"2023-04-05T13:22:21.809204Z","shell.execute_reply.started":"2023-04-05T13:22:21.317985Z","shell.execute_reply":"2023-04-05T13:22:21.808034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## This is where we balance class weights\nfrom sklearn.utils import compute_class_weight\ntrain_classes = Y_train\nclass_weights = compute_class_weight(\n                                        class_weight = \"balanced\",\n                                        classes = np.unique(train_classes),\n                                        y = train_classes                                                    \n                                    )\nclass_weights = dict(zip(np.unique(train_classes), class_weights))\nclass_weights","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:22:21.810925Z","iopub.execute_input":"2023-04-05T13:22:21.812216Z","iopub.status.idle":"2023-04-05T13:22:21.823808Z","shell.execute_reply.started":"2023-04-05T13:22:21.812166Z","shell.execute_reply":"2023-04-05T13:22:21.822527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training the model","metadata":{}},{"cell_type":"code","source":"callback = tf.keras.callbacks.ModelCheckpoint(\n    filepath='our_cnn_best.h5',\n    monitor='val_binary_accuracy',\n    mode='max',\n    save_best_only=True, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:22:21.826184Z","iopub.execute_input":"2023-04-05T13:22:21.826662Z","iopub.status.idle":"2023-04-05T13:22:21.833819Z","shell.execute_reply.started":"2023-04-05T13:22:21.826622Z","shell.execute_reply":"2023-04-05T13:22:21.832635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(\n    x_train,\n    y_train,\n    epochs = 10,\n    batch_size=20,\n    validation_data = (x_test,y_test),\n    class_weight= class_weights,\n    callbacks = callback\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:22:21.835482Z","iopub.execute_input":"2023-04-05T13:22:21.836435Z","iopub.status.idle":"2023-04-05T13:22:41.897316Z","shell.execute_reply.started":"2023-04-05T13:22:21.836393Z","shell.execute_reply":"2023-04-05T13:22:41.896285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_cnn = load_model('/kaggle/working/our_cnn_best.h5', custom_objects={\"f1_score\": f1_score })\nbest_cnn.evaluate(x_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:22:41.900391Z","iopub.execute_input":"2023-04-05T13:22:41.90068Z","iopub.status.idle":"2023-04-05T13:22:42.829625Z","shell.execute_reply.started":"2023-04-05T13:22:41.900651Z","shell.execute_reply":"2023-04-05T13:22:42.828607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot confusion matrices for benchmark and transfer learning models\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\n\nplt.figure(figsize=(15, 5))\n\npreds = best_cnn.predict(x_test)\npreds = (preds >= 0.5).astype(np.int32)\n\ncm = confusion_matrix(y_test, preds)\ndf_cm = pd.DataFrame(cm, index=['LAA', 'CE'], columns=['LAA', 'CE'])\nplt.subplot(121)\nplt.title(\"Confusion matrix for our model\\n\")\nsns.heatmap(df_cm, annot=True, fmt=\"d\", cmap=\"YlGnBu\")\nplt.ylabel(\"Predicted\")\nplt.xlabel(\"Actual\")","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:22:42.831174Z","iopub.execute_input":"2023-04-05T13:22:42.831479Z","iopub.status.idle":"2023-04-05T13:22:43.446931Z","shell.execute_reply.started":"2023-04-05T13:22:42.83145Z","shell.execute_reply":"2023-04-05T13:22:43.445869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test1=[]\nfor i in test_df['file_path']:\n    x1=preprocess(i)\n    test1.append(x1)\n    print(i)\n    \ntest1=np.array(test1)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:22:43.448697Z","iopub.execute_input":"2023-04-05T13:22:43.44945Z","iopub.status.idle":"2023-04-05T13:23:02.889065Z","shell.execute_reply.started":"2023-04-05T13:22:43.449408Z","shell.execute_reply":"2023-04-05T13:23:02.887841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn_pred=model.predict(test1)\ncnn_pred","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:23:02.89073Z","iopub.execute_input":"2023-04-05T13:23:02.891431Z","iopub.status.idle":"2023-04-05T13:23:03.196354Z","shell.execute_reply.started":"2023-04-05T13:23:02.891387Z","shell.execute_reply":"2023-04-05T13:23:03.193802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame(test_df[\"patient_id\"].copy())\nsub[\"CE\"] = cnn_pred\nsub[\"LAA\"] = 1- sub[\"CE\"]\n\nsub = sub.groupby(\"patient_id\").mean()\nsub = sub[[\"CE\", \"LAA\"]].round(6).reset_index()\nsub","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:23:03.203393Z","iopub.execute_input":"2023-04-05T13:23:03.207048Z","iopub.status.idle":"2023-04-05T13:23:03.243499Z","shell.execute_reply.started":"2023-04-05T13:23:03.20697Z","shell.execute_reply":"2023-04-05T13:23:03.242498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index = False)\n!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2023-04-05T13:23:03.245678Z","iopub.execute_input":"2023-04-05T13:23:03.2461Z","iopub.status.idle":"2023-04-05T13:23:04.405288Z","shell.execute_reply.started":"2023-04-05T13:23:03.24606Z","shell.execute_reply":"2023-04-05T13:23:04.40403Z"},"trusted":true},"execution_count":null,"outputs":[]}]}