{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Train simple CNN classifier\nHere we train a CNN to predict cancer grade because on tiled histological images. See the following notebook for the generation of these images:  \nhttps://www.kaggle.com/lvulliard/crop-images","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport shutil\n\n# There are two ways to load the data from the PANDA dataset:\n# Option 1: Load images using openslide\nimport openslide\n# Option 2: Load images using skimage (requires that tifffile is installed)\nimport skimage.io\n\n# General packages\nimport pandas as pd\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport PIL\nfrom IPython.display import Image, display\nfrom collections import Counter\n\nimport cv2\nimport skimage.io\nfrom tqdm.notebook import tqdm\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import MultiLabelBinarizer\n\nfrom keras.optimizers import Adam\nfrom keras.preprocessing.image import ImageDataGenerator","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\ntf.test.is_gpu_available()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Install Efficient-net\nResources collected here: https://www.kaggle.com/prateekagnihotri/efficientnet-keras-train-qwk-loss-augmentation\n","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install -qq /kaggle/input/sequencedata/efficientnet-1.1.0-py3-none-any.whl","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Set directories and parameters","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Location of the training images\ndataDir = '/kaggle/input/crop-images/cropped_train_images/'\n\n# Location of training labels\ntrainLabels = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/train.csv').set_index('image_id')\ntestDF = pd.read_csv('/kaggle/input/prostate-cancer-grade-assessment/test.csv').set_index('image_id')\n\n# Output cropped images\ncropDir = '/kaggle/working/cropped_train_images/'\n\ninputShape = (224, 224, 3)\nepochs = 100","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# How many train objects should be included in one batch (higher = faster but less accurate)\n# Take care that the batch size is smaller than the amount of total images analyzed\nbatchSize = 256\nINIT_LR = 1e-4","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"## Data generators","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"trainDatagen = ImageDataGenerator(rotation_range=30, width_shift_range=0.1,height_shift_range=0.1, validation_split = 0.1,\n                                  zoom_range=0.2, horizontal_flip=True, fill_mode=\"nearest\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainDF = pd.DataFrame(list(zip(trainLabels.index + \".png\", trainLabels.isup_grade.astype(str))), \n               columns =['x_col', 'y_col']) \n\n# Uncomment the following if the assumption needs to be re-checked\n# for x in trainDF.x_col:\n#     assert x in os.listdir(dataDir)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainGenerator = trainDatagen.flow_from_dataframe(\n    trainDF, x_col=\"x_col\", y_col=\"y_col\",\n    directory=dataDir,  # this is the target directory\n    batch_size=batchSize,\n    class_mode = \"sparse\",\n    subset=\"training\",\n    target_size=(inputShape[0], inputShape[1]),\n    color_mode='rgb')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valGenerator = trainDatagen.flow_from_dataframe(\n    trainDF, x_col=\"x_col\", y_col=\"y_col\",\n    directory=dataDir,  # this is the target directory\n    batch_size=batchSize,\n    class_mode = \"sparse\",\n    subset=\"validation\",\n    target_size=(inputShape[0], inputShape[1]),\n    color_mode='rgb')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## EfficientNet model\nSee https://www.kaggle.com/hassanamin/transfer-learning-vgg16-examples-using-tensorflow#Specify-the-Model","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import efficientnet.keras as efn \nfrom keras.models import Sequential\nfrom keras.layers import Dense, Flatten, GlobalAveragePooling2D","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"numClasses = len(set(trainLabels.isup_grade))\nweightFile = \"/kaggle/input/efficientnet-weights-for-keras/noisy-student/notop/efficientnet-b1_noisy-student_notop.h5\"\n\nmyModel = Sequential()\nmyModel.add(efn.EfficientNetB1(include_top=False, pooling='avg', weights=weightFile))\nmyModel.add(Dense(numClasses, activation='softmax'))\n\n# Say not to train first layer (pre-trainerd model)\nmyModel.layers[0].trainable = False","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"myModel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Optimaztion function\nopt = Adam(lr=INIT_LR, decay=INIT_LR / epochs)\n\nmyModel.compile(loss=\"sparse_categorical_crossentropy\",\n              optimizer=opt,\n              metrics=[\"binary_accuracy\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"H = myModel.fit_generator(trainGenerator,\n                        steps_per_epoch=len(trainLabels.index) // batchSize,\n                        epochs=epochs, \n                        validation_data=valGenerator,\n                        validation_steps=1,\n                        verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Save model\nmyModel.save('/kaggle/working/B1_'+str(epochs)+'.model')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}