{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip -q install tensorflow==2.3.0","metadata":{"execution":{"iopub.status.busy":"2021-09-28T01:05:20.766087Z","iopub.status.idle":"2021-09-28T01:05:20.76665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Basics / Data manipulation\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nimport zipfile\nimport os\n\n# Visualization\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport cv2\nimport skimage.io\n\n# ML\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2021-09-30T00:56:04.72783Z","iopub.execute_input":"2021-09-30T00:56:04.728206Z","iopub.status.idle":"2021-09-30T00:56:12.951537Z","shell.execute_reply.started":"2021-09-30T00:56:04.728171Z","shell.execute_reply":"2021-09-30T00:56:12.950176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split the data\nWe begin by splitting the data, we will allocate **80%** of the images for the training and** 20%** for the training on the 10k+ of .tiff images","metadata":{}},{"cell_type":"code","source":"# Folder paths\nTRAIN = '../input/prostate-cancer-grade-assessment/train_images/'\nMASKS = '../input/prostate-cancer-grade-assessment/train_label_masks/'\nOUT_TRAIN = 'train.zip'\nOUT_TEST = 'test.zip'\nOUT_MASKS_TRAIN = 'masks_train.zip'\nOUT_MASKS_TEST = 'masks_test.zip'\n\nBASE_FOLDER = \"/kaggle/input/prostate-cancer-grade-assessment/\"\n!ls {BASE_FOLDER}","metadata":{"execution":{"iopub.status.busy":"2021-09-30T00:56:17.787725Z","iopub.execute_input":"2021-09-30T00:56:17.787982Z","iopub.status.idle":"2021-09-30T00:56:18.061725Z","shell.execute_reply.started":"2021-09-30T00:56:17.787957Z","shell.execute_reply":"2021-09-30T00:56:18.06035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(BASE_FOLDER+\"train.csv\")\ntest = pd.read_csv(BASE_FOLDER+\"test.csv\")\nsub = pd.read_csv(BASE_FOLDER+\"sample_submission.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-30T00:59:03.156001Z","iopub.execute_input":"2021-09-30T00:59:03.156247Z","iopub.status.idle":"2021-09-30T00:59:03.23349Z","shell.execute_reply.started":"2021-09-30T00:59:03.156221Z","shell.execute_reply":"2021-09-30T00:59:03.232438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking for all the \"negative\" labels in the label of gleason_score\ntrain[train['gleason_score'] == 'negative']","metadata":{"execution":{"iopub.status.busy":"2021-09-30T00:59:07.663317Z","iopub.execute_input":"2021-09-30T00:59:07.66359Z","iopub.status.idle":"2021-09-30T00:59:07.692804Z","shell.execute_reply.started":"2021-09-30T00:59:07.663563Z","shell.execute_reply":"2021-09-30T00:59:07.691024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Deleting from the dataset a mislabeled row and converting the \"negative\" labels to \"0+0\" in order to have an standard\ntrain.drop([7273],inplace=True)\ntrain['gleason_score'] = train['gleason_score'].apply(lambda x: \"0+0\" if x == \"negative\" else x)","metadata":{"execution":{"iopub.status.busy":"2021-09-30T00:59:14.321911Z","iopub.execute_input":"2021-09-30T00:59:14.322153Z","iopub.status.idle":"2021-09-30T00:59:14.332207Z","shell.execute_reply.started":"2021-09-30T00:59:14.32213Z","shell.execute_reply":"2021-09-30T00:59:14.330956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = np.array(train) # Converting the DataFrame to an array to take the column\nlabels = data[:, 3] # Labels of interest (GLEASON SCORE)\nlabels","metadata":{"execution":{"iopub.status.busy":"2021-09-30T00:59:18.141498Z","iopub.execute_input":"2021-09-30T00:59:18.141737Z","iopub.status.idle":"2021-09-30T00:59:18.149871Z","shell.execute_reply.started":"2021-09-30T00:59:18.141715Z","shell.execute_reply":"2021-09-30T00:59:18.148762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = data[:, :3] # Features of interest (ID, PROVIDER, ISUP GRADE)\nfeatures","metadata":{"execution":{"iopub.status.busy":"2021-09-30T00:59:23.10158Z","iopub.execute_input":"2021-09-30T00:59:23.101922Z","iopub.status.idle":"2021-09-30T00:59:23.109623Z","shell.execute_reply.started":"2021-09-30T00:59:23.101893Z","shell.execute_reply":"2021-09-30T00:59:23.108093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = features\ny = labels","metadata":{"execution":{"iopub.status.busy":"2021-09-30T00:59:29.275763Z","iopub.execute_input":"2021-09-30T00:59:29.276Z","iopub.status.idle":"2021-09-30T00:59:29.280187Z","shell.execute_reply.started":"2021-09-30T00:59:29.275977Z","shell.execute_reply":"2021-09-30T00:59:29.27936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2021-09-30T00:59:33.816184Z","iopub.execute_input":"2021-09-30T00:59:33.816442Z","iopub.status.idle":"2021-09-30T00:59:33.820887Z","shell.execute_reply.started":"2021-09-30T00:59:33.816416Z","shell.execute_reply":"2021-09-30T00:59:33.819905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.20, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2021-09-30T00:59:38.024915Z","iopub.execute_input":"2021-09-30T00:59:38.025351Z","iopub.status.idle":"2021-09-30T00:59:38.032234Z","shell.execute_reply.started":"2021-09-30T00:59:38.025326Z","shell.execute_reply":"2021-09-30T00:59:38.031346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Distribution of the data\n\nEDA on the splitted data","metadata":{}},{"cell_type":"code","source":"X_train = pd.DataFrame(X_train, columns=[\"image_id\", \"data_provider\", \"isup_grade\"])\nX_train[\"gleason_score\"] = y_train\n\nX_test = pd.DataFrame(X_test, columns=[\"image_id\", \"data_provider\", \"isup_grade\"])\nX_test[\"gleason_score\"] = y_test\n\ntrain_eda = X_train.groupby(\"gleason_score\").count()[\"image_id\"].reset_index().sort_values(by=\"image_id\", ascending=False)\ntrain_eda.style.background_gradient(cmap=\"Greens\")","metadata":{"execution":{"iopub.status.busy":"2021-09-30T00:59:43.806297Z","iopub.execute_input":"2021-09-30T00:59:43.806727Z","iopub.status.idle":"2021-09-30T00:59:43.950862Z","shell.execute_reply.started":"2021-09-30T00:59:43.806702Z","shell.execute_reply":"2021-09-30T00:59:43.949943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_eda = X_test.groupby(\"gleason_score\").count()[\"image_id\"].reset_index().sort_values(by=\"image_id\", ascending=False)\ntest_eda.style.background_gradient(cmap=\"Blues\")","metadata":{"execution":{"iopub.status.busy":"2021-09-24T01:57:44.46033Z","iopub.execute_input":"2021-09-24T01:57:44.46065Z","iopub.status.idle":"2021-09-24T01:57:44.493249Z","shell.execute_reply.started":"2021-09-24T01:57:44.460617Z","shell.execute_reply":"2021-09-24T01:57:44.492417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\n\nfig = go.Figure(data=[\n    go.Bar(name=\"Test\", x=train_eda[\"gleason_score\"], y=train_eda[\"image_id\"]),\n    go.Bar(name=\"Train\", x=test_eda[\"gleason_score\"], y=test_eda[\"image_id\"]),\n])\n\n# Change the bar mode\nfig.update_layout(barmode='group')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-28T01:05:20.678534Z","iopub.execute_input":"2021-09-28T01:05:20.678828Z","iopub.status.idle":"2021-09-28T01:05:20.764511Z","shell.execute_reply.started":"2021-09-28T01:05:20.678799Z","shell.execute_reply":"2021-09-28T01:05:20.76319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n\ndf = train_eda\nfig = px.pie(df, values='image_id', names='gleason_score')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-24T01:57:52.173226Z","iopub.execute_input":"2021-09-24T01:57:52.173521Z","iopub.status.idle":"2021-09-24T01:57:52.580381Z","shell.execute_reply.started":"2021-09-24T01:57:52.173493Z","shell.execute_reply":"2021-09-24T01:57:52.579788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = test_eda\nfig = px.pie(df, values='image_id', names='gleason_score')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-24T01:58:12.092121Z","iopub.execute_input":"2021-09-24T01:58:12.092414Z","iopub.status.idle":"2021-09-24T01:58:12.548231Z","shell.execute_reply.started":"2021-09-24T01:58:12.092383Z","shell.execute_reply":"2021-09-24T01:58:12.547499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Proving that 80% of the data goes to the training and 20% goes to the training\n\nIn the below image we can see the representation of the distribution of the dataset","metadata":{}},{"cell_type":"code","source":"labels = 'Training Images', 'Testing Images'\nsizes_features = [len(X_train), len(X_test)]\n# sizes_labels = [len(y_train), len(y_test)]\n\nfig, ax = plt.subplots(figsize=(30,7))\n\nax.pie(sizes_features, labels=labels, autopct='%1.1f%%',\n          shadow=True, startangle=60)\nax.axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle\nax.set_title(f\"Distribution of the dataset\\n Total Images - {(len(X) / len(X) * 100)}%: {len(X)}\\n Testing images - {(len(X_train) / len(X) * 100)}%: {len(X_train)} \\n Training images - {(len(X_test) / len(X) * 100)}%: {len(X_test)}\"\n                                                                          ,weight=\"bold\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-24T01:58:16.88936Z","iopub.execute_input":"2021-09-24T01:58:16.889685Z","iopub.status.idle":"2021-09-24T01:58:16.992386Z","shell.execute_reply.started":"2021-09-24T01:58:16.889652Z","shell.execute_reply":"2021-09-24T01:58:16.991454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the datasets\nX_train = pd.DataFrame(X_train, columns=[\"image_id\", \"data_provider\", \"isup_grade\", \"gleason_score\"])\nX_test = pd.DataFrame(X_test, columns=[\"image_id\", \"data_provider\", \"isup_grade\", \"gleason_score\"])\n\nX_train.to_csv(\"./training.csv\")\nX_test.to_csv(\"./testing.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-09-30T00:59:58.430474Z","iopub.execute_input":"2021-09-30T00:59:58.430723Z","iopub.status.idle":"2021-09-30T00:59:58.646095Z","shell.execute_reply.started":"2021-09-30T00:59:58.430698Z","shell.execute_reply":"2021-09-30T00:59:58.645059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SIZE_IMG = 112\nN = 16\ndef tile(img, mask):\n    result = []\n    shape = img.shape\n    pad0,pad1 = (SIZE_IMG - shape[0]%SIZE_IMG)%SIZE_IMG, (SIZE_IMG - shape[1]%SIZE_IMG)%SIZE_IMG\n    img = np.pad(img, [[pad0//2, pad0-pad0//2], [pad1//2, pad1 - pad1//2],[0,0]],\n                constant_values=255)\n    mask = np.pad(mask,[[pad0//2, pad0-pad0//2], [pad1//2,pad1-pad1//2], [0,0]],\n                constant_values=0)\n    img = img.reshape(img.shape[0]//SIZE_IMG, SIZE_IMG, img.shape[1]//SIZE_IMG,SIZE_IMG, 3)\n    img = img.transpose(0,2,1,3,4).reshape(-1, SIZE_IMG,SIZE_IMG,3)\n    mask = mask.reshape(mask.shape[0]//SIZE_IMG, SIZE_IMG,mask.shape[1]//SIZE_IMG, SIZE_IMG, 3)\n    mask = mask.transpose(0, 2, 1, 3, 4).reshape(-1, SIZE_IMG,SIZE_IMG, 3)\n    if len(img) < N:\n        mask = np.pad(mask, [[0, N-len(img)], [0, 0], [0, 0],[0, 0]], constant_values=0)\n        img = np.pad(img, [[0, N-len(img)],[0, 0],[0, 0], [0, 0]], constant_values=255)\n    idxs = np.argsort(img.reshape(img.shape[0], -1).sum(-1))[: N]\n    img = img[idxs]\n    mask = mask[idxs]\n    for i in range(len(img)):\n        result.append({'img':img[i], 'mask':mask[i], 'idx':i})\n    return result","metadata":{"execution":{"iopub.status.busy":"2021-09-30T01:00:05.186071Z","iopub.execute_input":"2021-09-30T01:00:05.186358Z","iopub.status.idle":"2021-09-30T01:00:05.199783Z","shell.execute_reply.started":"2021-09-30T01:00:05.186335Z","shell.execute_reply":"2021-09-30T01:00:05.198853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Final appeareance\n\nHere, we just display a preview of an specific image on how it will look at the end","metadata":{}},{"cell_type":"code","source":"train_dataset = pd.read_csv(\"./training.csv\", usecols=[\"image_id\", \"data_provider\", \"isup_grade\", \"gleason_score\"])\ntest_dataset = pd.read_csv(\"./testing.csv\", usecols=[\"image_id\", \"data_provider\", \"isup_grade\", \"gleason_score\"])\n\nf, ax = plt.subplots(4,4, figsize=(10, 10))\n\n# Mapping to the original dataset\nimg = skimage.io.MultiImage(os.path.join(TRAIN,\"5801d2195cdcd8d336e8fc097c51a788\"+'.tiff'))[1]\nmask = skimage.io.MultiImage(os.path.join(MASKS,\"5801d2195cdcd8d336e8fc097c51a788\"+'_mask.tiff'))[1]\ntiles = tile(img, mask)\nfor t in range(len(tiles)):\n    ax[t//4, t%4].imshow(tiles[t][\"img\"]) # Displaying Image    \n    ax[t//4, t%4].axis('off')      ","metadata":{"execution":{"iopub.status.busy":"2021-09-30T01:00:10.873199Z","iopub.execute_input":"2021-09-30T01:00:10.873606Z","iopub.status.idle":"2021-09-30T01:00:11.831873Z","shell.execute_reply.started":"2021-09-30T01:00:10.873577Z","shell.execute_reply":"2021-09-30T01:00:11.83034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Function to concatenate the 16 tiles in one image\nThe function ```concant_tile()``` concantenates the 16 tiles in one single image, which will be saved later on.","metadata":{}},{"cell_type":"code","source":"id_train = train_dataset[\"image_id\"][163]\nid_test = test_dataset[\"image_id\"][163]\n\n# Testing the function\ndef concat_tile(im_list_2d):\n    return cv2.vconcat([cv2.hconcat(im_list_h) for im_list_h in im_list_2d])\n\ndef mosaic(tiles):\n\n    im1 = tiles[0][\"img\"]\n    im2 = tiles[1][\"img\"]\n    im3 = tiles[2][\"img\"]\n    im4 = tiles[3][\"img\"]\n\n    im5 = tiles[4][\"img\"]\n    im6 = tiles[5][\"img\"]\n    im7 = tiles[6][\"img\"]\n    im8 = tiles[7][\"img\"]\n\n    im9 = tiles[8][\"img\"]\n    im10 = tiles[9][\"img\"]\n    im11 = tiles[10][\"img\"]\n    im12 = tiles[11][\"img\"]\n\n    im13 = tiles[12][\"img\"]\n    im14 = tiles[13][\"img\"]\n    im15 = tiles[14][\"img\"]\n    im16 = tiles[15][\"img\"]\n\n    im_tile = concat_tile([[im1, im2, im3, im4],\n                           [im5, im6, im7, im8],\n                           [im9, im10, im11, im12],\n                           [im13, im14, im15, im16]])\n    return im_tile\n\nimg = skimage.io.MultiImage(os.path.join(TRAIN, f\"{id_train}.tiff\"))[1]\nmask = skimage.io.MultiImage(os.path.join(MASKS, f\"{id_train}_mask.tiff\"))[1]\ntiles = tile(img, mask)\n\nmosaic_img = mosaic(tiles)\nplt.title(f\"ID: {id_train}\")\nim=plt.imshow(mosaic_img)","metadata":{"execution":{"iopub.status.busy":"2021-09-30T01:02:59.316292Z","iopub.execute_input":"2021-09-30T01:02:59.316524Z","iopub.status.idle":"2021-09-30T01:02:59.765769Z","shell.execute_reply.started":"2021-09-30T01:02:59.316502Z","shell.execute_reply":"2021-09-30T01:02:59.764573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Function to generate the dataset\n\nIn this function we try to generate a dataset for the training and testing images that will work for training and predicting the ISUP score. We do it like this:\n\n* Iterate through the train and test dataset\n    * Map the for both the train and the test dataset to the base folder\n    * Zip the 16 subimages\n    * Save the 16 subimages in their correspondant GLEASON_SCORE folder","metadata":{}},{"cell_type":"code","source":"train_IDs = train_dataset[\"image_id\"]\ntest_IDs = test_dataset[\"image_id\"]\n\nnot_found_train = []\nnot_found_test = []\n\ndef generate_dataset(ids, train=True):\n    if train:\n        x_tot,x2_tot = [], []\n        with zipfile.ZipFile(OUT_TRAIN, 'w') as img_out,\\\n         zipfile.ZipFile(OUT_MASKS_TRAIN, 'w') as mask_out:\n            for gleason_score, id in enumerate(tqdm(ids)):\n                try:\n                    img = skimage.io.MultiImage(os.path.join(TRAIN,id+'.tiff'))[1]\n                    mask = skimage.io.MultiImage(os.path.join(MASKS,id+'_mask.tiff'))[1]\n                    tiles = tile(img,mask)                    \n                    img = mosaic(tiles)\n                    \n                    x_tot.append((img/255.0).reshape(-1,3).mean(0))\n                    x2_tot.append(((img/255.0)**2).reshape(-1,3).mean(0))\n                    # If read with PIL RGB turns into BGR\n                    img = cv2.imencode('.png',cv2.cvtColor(img, cv2.COLOR_RGB2BGR))[1]\n                    # Uncomment to classify by ISUP GRADE \n                    # img_out.writestr(f'train/ISUP_GRADE_{train_dataset[\"isup_grade\"][isup_grade]}/{id}_{idx}.png', img)\n                    img_out.writestr(f'train/GLEASON_SCORE_{train_dataset[\"gleason_score\"][gleason_score]}/{id}.png', img)\n                except Exception as e:\n                    not_found_train.append(id)\n        print(f\"INFO: Not images found in train: {len(not_found_train)}\")\n        \n    if not train: \n        x_tot,x2_tot = [], []\n        with zipfile.ZipFile(OUT_TEST, 'w') as img_out,\\\n         zipfile.ZipFile(OUT_MASKS_TEST, 'w') as mask_out:\n            for gleason_score, id in enumerate(tqdm(ids)):\n                try:\n                    img = skimage.io.MultiImage(os.path.join(TRAIN,id+'.tiff'))[1]\n                    mask = skimage.io.MultiImage(os.path.join(MASKS,id+'_mask.tiff'))[1]\n                    tiles = tile(img,mask)\n                    img = mosaic(tiles)\n                    \n                    x_tot.append((img/255.0).reshape(-1,3).mean(0))\n                    x2_tot.append(((img/255.0)**2).reshape(-1,3).mean(0)) \n                    # If read with PIL RGB turns into BGR\n                    img = cv2.imencode('.png',cv2.cvtColor(img, cv2.COLOR_RGB2BGR))[1]\n                    # Uncomment to classify by ISUP GRADE \n                    # img_out.writestr(f'test/ISUP_GRADE_{train_dataset[\"isup_grade\"][isup_grade]}/{id}_{idx}.png', img)\n                    img_out.writestr(f'test/GLEASON_SCORE_{test_dataset[\"gleason_score\"][gleason_score]}/{id}.png', img)\n                except Exception as e:\n                    not_found_test.append(id)\n\n        print(f\"INFO: Not images found in test: {len(not_found_test)}\")","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:36.961195Z","iopub.execute_input":"2021-08-28T06:28:36.961483Z","iopub.status.idle":"2021-08-28T06:28:36.983391Z","shell.execute_reply.started":"2021-08-28T06:28:36.961453Z","shell.execute_reply":"2021-08-28T06:28:36.982607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"generate_dataset(train_IDs, train=True)\ngenerate_dataset(test_IDs, train=False)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:40.523188Z","iopub.execute_input":"2021-08-28T06:28:40.523471Z","iopub.status.idle":"2021-08-28T06:28:49.385651Z","shell.execute_reply.started":"2021-08-28T06:28:40.523441Z","shell.execute_reply":"2021-08-28T06:28:49.384815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Images loss\n\nUnfortunately, some images were not able to be processed for some reason. Thankfully, they are not enough to disturb the training and testing.","metadata":{}},{"cell_type":"code","source":"labels = \"Training Images\", \"Testing Images\", \"Loss\"\nsizes_features = [len(X_train), len(X_test), len(not_found_train) + len(not_found_test)]\n# sizes_labels = [len(X_train), len(X_test), 20]\n\nfig, ax = plt.subplots(figsize=(30,7))\n\nax.pie(sizes_features, labels=labels, autopct='%1.1f%%',\n          shadow=True, startangle=60)\nax.axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle\nax.set_title(f\"Distribution of the dataset\\n Total Images - {(len(X) / len(X) * 100)}%: {len(X)}\\n Testing images - {(len(X_train) / len(X) * 100)}%: {len(X_train)} \\n Training images - {(len(X_test) / len(X) * 100)}%: {len(X_test)}\\n\\n Loss - : {len(not_found_train) + len(not_found_test)} / {len(X)} images\", \n                                                            weight=\"bold\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:50.853969Z","iopub.execute_input":"2021-08-28T06:28:50.854279Z","iopub.status.idle":"2021-08-28T06:28:50.965824Z","shell.execute_reply.started":"2021-08-28T06:28:50.854248Z","shell.execute_reply":"2021-08-28T06:28:50.964968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Making sure the not found images DO NOT affect the whole dataset\n\nSince there are 100 images not found (80 images for training and 20 for testing) we want to make sure this does not affect at the end. Given that there is a possibility that the 80 images not found may belong to the classes with the less images provided by the original dataset","metadata":{}},{"cell_type":"code","source":"not_found_train_eda = []\nif not_found_train:\n    for not_found in not_found_train:\n        not_found_train_eda.append(train_dataset[train_dataset[\"image_id\"] == not_found])\n    not_found_train_eda = pd.concat(not_found_train_eda)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:51.206999Z","iopub.execute_input":"2021-08-28T06:28:51.207315Z","iopub.status.idle":"2021-08-28T06:28:51.212643Z","shell.execute_reply.started":"2021-08-28T06:28:51.207282Z","shell.execute_reply":"2021-08-28T06:28:51.21176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"not_found_test_eda = []\nif not_found_test:\n    for not_found in not_found_test:\n        not_found_test_eda.append(test_dataset[test_dataset[\"image_id\"] == not_found])\n    not_found_test_eda = pd.concat(not_found_test_eda)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:51.49546Z","iopub.execute_input":"2021-08-28T06:28:51.495777Z","iopub.status.idle":"2021-08-28T06:28:51.502789Z","shell.execute_reply.started":"2021-08-28T06:28:51.495738Z","shell.execute_reply":"2021-08-28T06:28:51.499561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if not_found_train_eda.empty:\nnot_found_train_eda = not_found_train_eda.groupby('gleason_score').count()['image_id'].reset_index().sort_values(by='image_id', ascending=False)\nnot_found_train_eda.style.background_gradient(cmap='Greens')","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:51.710553Z","iopub.execute_input":"2021-08-28T06:28:51.710873Z","iopub.status.idle":"2021-08-28T06:28:52.336072Z","shell.execute_reply.started":"2021-08-28T06:28:51.710839Z","shell.execute_reply":"2021-08-28T06:28:52.334242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if not not_found_test_eda.empty:\nnot_found_test_eda = not_found_test_eda.groupby('gleason_score').count()['image_id'].reset_index().sort_values(by='image_id', ascending=False)\nnot_found_test_eda.style.background_gradient(cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:52.337132Z","iopub.status.idle":"2021-08-28T06:28:52.337444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if not_found_train_eda or not_found_test_eda:\nfig = go.Figure(data=[\n    go.Bar(name=\"Not found test\", x=not_found_train_eda[\"gleason_score\"], y=not_found_train_eda[\"image_id\"]),\n    go.Bar(name=\"Not found train\", x=not_found_test_eda[\"gleason_score\"], y=not_found_test_eda[\"image_id\"])\n])\n\n# Change the bar mode\nfig.update_layout(barmode='group')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:52.440442Z","iopub.execute_input":"2021-08-28T06:28:52.44086Z","iopub.status.idle":"2021-08-28T06:28:52.471463Z","shell.execute_reply.started":"2021-08-28T06:28:52.440819Z","shell.execute_reply":"2021-08-28T06:28:52.470078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if not_found_train_eda or not_found_test_eda:\nfig = go.Figure(data=[\n    go.Bar(name=\"Not found test\", x=not_found_train_eda[\"gleason_score\"], y=not_found_train_eda[\"image_id\"]),\n    go.Bar(name=\"Not found train\", x=not_found_test_eda[\"gleason_score\"], y=not_found_test_eda[\"image_id\"]),\n    go.Bar(name=\"Found test\", x=train_eda[\"gleason_score\"], y=train_eda[\"image_id\"]),\n    go.Bar(name=\"Found train\", x=test_eda[\"gleason_score\"], y=test_eda[\"image_id\"])\n])\n\n# Change the bar mode\nfig.update_layout(barmode='group')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:52.746175Z","iopub.execute_input":"2021-08-28T06:28:52.746462Z","iopub.status.idle":"2021-08-28T06:28:52.77518Z","shell.execute_reply.started":"2021-08-28T06:28:52.746433Z","shell.execute_reply":"2021-08-28T06:28:52.773783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Distribution of the loss images\nIn the next pie charts, it is shown the loss in the images for both the testing and training datasets ","metadata":{}},{"cell_type":"code","source":"# if not_found_train_eda:\ndf = not_found_train_eda\nfig = px.pie(df, values='image_id', names='gleason_score')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:53.583846Z","iopub.execute_input":"2021-08-28T06:28:53.584137Z","iopub.status.idle":"2021-08-28T06:28:53.61659Z","shell.execute_reply.started":"2021-08-28T06:28:53.584107Z","shell.execute_reply":"2021-08-28T06:28:53.615341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if not_found_test_eda:\ndf = not_found_test_eda\nfig = px.pie(df, values='image_id', names='gleason_score')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:53.909608Z","iopub.execute_input":"2021-08-28T06:28:53.909924Z","iopub.status.idle":"2021-08-28T06:28:53.941968Z","shell.execute_reply.started":"2021-08-28T06:28:53.909893Z","shell.execute_reply":"2021-08-28T06:28:53.940721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking if GPU is being used","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:54.482558Z","iopub.execute_input":"2021-08-28T06:28:54.482867Z","iopub.status.idle":"2021-08-28T06:28:54.486996Z","shell.execute_reply.started":"2021-08-28T06:28:54.482835Z","shell.execute_reply":"2021-08-28T06:28:54.485984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection\n    print(\"Running on TPU \", tpu.cluster_spec().as_dict()[\"worker\"])\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept ValueError:\n    print(\"Not connected to a TPU runtime. Using CPU/GPU strategy\")\n    strategy = tf.distribute.MirroredStrategy()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:54.733069Z","iopub.execute_input":"2021-08-28T06:28:54.733369Z","iopub.status.idle":"2021-08-28T06:28:55.528499Z","shell.execute_reply.started":"2021-08-28T06:28:54.733339Z","shell.execute_reply":"2021-08-28T06:28:55.527665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Unzipping the images\n!unzip -q -o train.zip\n!unzip -q -o test.zip","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:55.533766Z","iopub.execute_input":"2021-08-28T06:28:55.536509Z","iopub.status.idle":"2021-08-28T06:28:57.008315Z","shell.execute_reply.started":"2021-08-28T06:28:55.536466Z","shell.execute_reply":"2021-08-28T06:28:57.007222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Set-up EfficientNet\n\nIn order to train the CNN we will be using a powerful CNN that performs quiet well for training images. It comes in different \"flavours\" going from **EfficientNetB0** to **EfficientNetB7** (the greater the B#, the more input image size it can handle). \n\n","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import models\nfrom tensorflow.keras import layers\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom sklearn import model_selection\nfrom tensorflow.keras import optimizers\n#Use this to check if the GPU is configured correctly\nfrom tensorflow.python.client import device_lib\nprint(device_lib.list_local_devices())\n\n\nfrom tensorflow.keras.applications import EfficientNetB0","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:57.010374Z","iopub.execute_input":"2021-08-28T06:28:57.010621Z","iopub.status.idle":"2021-08-28T06:28:57.024629Z","shell.execute_reply.started":"2021-08-28T06:28:57.010594Z","shell.execute_reply":"2021-08-28T06:28:57.023613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Configuration of the EfficientNet\nconv_base = EfficientNetB0(weights=\"imagenet\", include_top=False, input_shape=(224, 224, 3))","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:28:57.026233Z","iopub.execute_input":"2021-08-28T06:28:57.026942Z","iopub.status.idle":"2021-08-28T06:29:00.046111Z","shell.execute_reply.started":"2021-08-28T06:28:57.026843Z","shell.execute_reply":"2021-08-28T06:29:00.045359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model \nThe model will have the follow configuration:\n______________\n1st layer: EfficientNetNB0 (224, 224, 3) input images\n______________\n2nd layer: GlobalMaxPooling2D\n______________\n3rd layer: Dropout with learning rate = 2e-5\n______________\n4th layer: Denser layer x 6 that will classify the image","metadata":{}},{"cell_type":"code","source":"model = models.Sequential()\nmodel.add(conv_base)\nmodel.add(layers.GlobalMaxPooling2D(name=\"gap\"))\n# Avoid overfitting\nmodel.add(layers.Dropout(rate=0.5))\nmodel.add(layers.Dense(10, activation=\"softmax\", name=\"fc_out\"))\nconv_base.trainable = True\n\nmodel.compile(\n    loss=\"categorical_crossentropy\",\n    optimizer=optimizers.RMSprop(lr=2e-5),\n    metrics=[\"acc\"],\n)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:30:09.440859Z","iopub.execute_input":"2021-08-28T06:30:09.441158Z","iopub.status.idle":"2021-08-28T06:30:10.559454Z","shell.execute_reply.started":"2021-08-28T06:30:09.441128Z","shell.execute_reply":"2021-08-28T06:30:10.558526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:30:11.282048Z","iopub.execute_input":"2021-08-28T06:30:11.282324Z","iopub.status.idle":"2021-08-28T06:30:11.311936Z","shell.execute_reply.started":"2021-08-28T06:30:11.282295Z","shell.execute_reply":"2021-08-28T06:30:11.310621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Augmentation\n\nBefore training, we preprocess a little bit the image, in order to have a better perfomance on the predictions","metadata":{}},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:30:12.409999Z","iopub.execute_input":"2021-08-28T06:30:12.41028Z","iopub.status.idle":"2021-08-28T06:30:12.414193Z","shell.execute_reply.started":"2021-08-28T06:30:12.41025Z","shell.execute_reply":"2021-08-28T06:30:12.413301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = plt.imread(\"./train/GLEASON_SCORE_0+0/000920ad0b612851f8e01bcc880d9b3d.png\")","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:30:12.827471Z","iopub.execute_input":"2021-08-28T06:30:12.827774Z","iopub.status.idle":"2021-08-28T06:30:12.860847Z","shell.execute_reply.started":"2021-08-28T06:30:12.827737Z","shell.execute_reply":"2021-08-28T06:30:12.859554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" # Creating an object that will contain all the changes that will\n # be performed randomly to the images to help the training be more robust\n image_gen = ImageDataGenerator(\n                                width_shift_range=0.1,\n                                height_shift_range=0.1,\n                                rescale=1/255,\n                                shear_range=0.2,\n                                zoom_range=0.2,\n                                horizontal_flip=True,\n                                fill_mode=\"nearest\"\n                                )","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:30:13.262946Z","iopub.execute_input":"2021-08-28T06:30:13.26325Z","iopub.status.idle":"2021-08-28T06:30:13.267813Z","shell.execute_reply.started":"2021-08-28T06:30:13.263221Z","shell.execute_reply":"2021-08-28T06:30:13.267016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(image_gen.random_transform(sample))","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:30:14.210477Z","iopub.execute_input":"2021-08-28T06:30:14.210777Z","iopub.status.idle":"2021-08-28T06:30:14.234589Z","shell.execute_reply.started":"2021-08-28T06:30:14.210739Z","shell.execute_reply":"2021-08-28T06:30:14.233396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Flowing through directories to see the classes and the number of images\nprint(image_gen.flow_from_directory(\"./train\"))\nprint(image_gen.flow_from_directory(\"./test\"))","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:30:14.628312Z","iopub.execute_input":"2021-08-28T06:30:14.628601Z","iopub.status.idle":"2021-08-28T06:30:14.839055Z","shell.execute_reply.started":"2021-08-28T06:30:14.628571Z","shell.execute_reply":"2021-08-28T06:30:14.838201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32\ntrain_image_gen = image_gen.flow_from_directory(\"./train\",\n                                                target_size=(224, 224),\n                                                batch_size=batch_size,\n                                                class_mode=\"categorical\")","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:30:15.169974Z","iopub.execute_input":"2021-08-28T06:30:15.170258Z","iopub.status.idle":"2021-08-28T06:30:15.279901Z","shell.execute_reply.started":"2021-08-28T06:30:15.170229Z","shell.execute_reply":"2021-08-28T06:30:15.278497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_image_gen = image_gen.flow_from_directory(\"./test\",\n                                                target_size=(224, 224),\n                                                batch_size=batch_size,\n                                                class_mode=\"categorical\")","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:30:16.883493Z","iopub.execute_input":"2021-08-28T06:30:16.883853Z","iopub.status.idle":"2021-08-28T06:30:16.992994Z","shell.execute_reply.started":"2021-08-28T06:30:16.883812Z","shell.execute_reply":"2021-08-28T06:30:16.992153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_image_gen.class_indices","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:30:19.277081Z","iopub.execute_input":"2021-08-28T06:30:19.277367Z","iopub.status.idle":"2021-08-28T06:30:19.287013Z","shell.execute_reply.started":"2021-08-28T06:30:19.27734Z","shell.execute_reply":"2021-08-28T06:30:19.286184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training the model\n\nNow, we traini the model with the 134592 images","metadata":{}},{"cell_type":"code","source":"NUMBER_OF_TRAINING_IMAGES = 8412\nNUMBER_OF_TESTING_IMAGES = 2103\nresults = model.fit(\n    train_image_gen,\n    steps_per_epoch=NUMBER_OF_TRAINING_IMAGES // batch_size,\n    epochs=140,\n    validation_data=test_image_gen,\n    validation_steps=NUMBER_OF_TESTING_IMAGES // batch_size,\n    verbose=1,\n    use_multiprocessing=True,\n    workers=4,\n)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:30:25.403016Z","iopub.execute_input":"2021-08-28T06:30:25.403318Z","iopub.status.idle":"2021-08-28T06:33:41.115851Z","shell.execute_reply.started":"2021-08-28T06:30:25.403289Z","shell.execute_reply":"2021-08-28T06:33:41.110082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the synaptic weights of the model\nmodel.save(\"./EfficientNet-model.h5\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Results","metadata":{}},{"cell_type":"code","source":"results_df = pd.DataFrame({\"epoch\":[i + 1 for i in range(len(results.history[\"acc\"]))], \"acc\":results.history[\"acc\"], \"val_acc\":results.history[\"val_acc\"], \"loss\":results.history[\"loss\"], \"val_loss\":results.history[\"val_loss\"]})\nresults_df","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:34:16.095518Z","iopub.execute_input":"2021-08-28T06:34:16.095845Z","iopub.status.idle":"2021-08-28T06:34:16.112827Z","shell.execute_reply.started":"2021-08-28T06:34:16.095807Z","shell.execute_reply":"2021-08-28T06:34:16.112086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hist_acc(hist):\n    plt.plot(hist.history[\"acc\"])\n    plt.plot(hist.history[\"val_acc\"])\n    plt.title(\"Model Accuracy\")\n    plt.ylabel(\"Accuracy\")\n    plt.xlabel(\"Epoch\")\n    plt.legend([\"Accuracy\", \"Validation Accuracy\"], loc=\"upper left\")\n    plt.show()\n\nplot_hist_acc(results)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:34:19.388151Z","iopub.execute_input":"2021-08-28T06:34:19.388425Z","iopub.status.idle":"2021-08-28T06:34:19.545833Z","shell.execute_reply.started":"2021-08-28T06:34:19.388396Z","shell.execute_reply":"2021-08-28T06:34:19.544035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hist_loss(hist):\n    plt.plot(hist.history[\"loss\"])\n    plt.plot(hist.history[\"val_loss\"])\n    plt.title(\"Model Loss\")\n    plt.ylabel(\"Accuracy\")\n    plt.xlabel(\"Epoch\")\n    plt.legend([\"Loss\", \"Validation Loss\"], loc=\"upper left\")\n    plt.show()\n\nplot_hist_loss(results)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:34:20.902954Z","iopub.execute_input":"2021-08-28T06:34:20.903283Z","iopub.status.idle":"2021-08-28T06:34:21.046242Z","shell.execute_reply.started":"2021-08-28T06:34:20.903254Z","shell.execute_reply":"2021-08-28T06:34:21.045382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\n","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:34:22.015449Z","iopub.execute_input":"2021-08-28T06:34:22.015831Z","iopub.status.idle":"2021-08-28T06:34:22.024415Z","shell.execute_reply.started":"2021-08-28T06:34:22.01579Z","shell.execute_reply":"2021-08-28T06:34:22.023457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_generator = ImageDataGenerator()\ntest_data_generator = test_generator.flow_from_directory(\n    \"./test\", # Put your path here\n    target_size=(224, 224),\n    batch_size=32,\n    shuffle=False)\ntest_steps_per_epoch = np.math.ceil(test_data_generator.samples / test_data_generator.batch_size)\n\npredictions = model.predict_generator(test_data_generator, steps=test_steps_per_epoch)\n# Get most likely class\npredicted_classes = np.argmax(predictions, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:50:51.543075Z","iopub.execute_input":"2021-08-28T06:50:51.543356Z","iopub.status.idle":"2021-08-28T06:50:51.923971Z","shell.execute_reply.started":"2021-08-28T06:50:51.543327Z","shell.execute_reply":"2021-08-28T06:50:51.922983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true_classes = test_data_generator.classes\nclass_labels = list(test_data_generator.class_indices.keys())   ","metadata":{"execution":{"iopub.status.busy":"2021-08-28T06:51:11.902611Z","iopub.execute_input":"2021-08-28T06:51:11.902906Z","iopub.status.idle":"2021-08-28T06:51:11.90843Z","shell.execute_reply.started":"2021-08-28T06:51:11.902876Z","shell.execute_reply":"2021-08-28T06:51:11.906078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"report = classification_report(true_classes, predicted_classes, target_names=class_labels)\nprint(report)   ","metadata":{"execution":{"iopub.status.busy":"2021-08-28T07:00:44.887441Z","iopub.execute_input":"2021-08-28T07:00:44.887721Z","iopub.status.idle":"2021-08-28T07:00:44.900243Z","shell.execute_reply.started":"2021-08-28T07:00:44.887691Z","shell.execute_reply":"2021-08-28T07:00:44.899298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Deleting image folders to avoid over-saturate the output\n!rm -r test\n!rm -r train","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}