{"cells":[{"metadata":{"_uuid":"cbd72bac-700a-4d9c-a04c-c49cddf9c1dd","_cell_guid":"dc71dbf0-7ede-4f17-ac0f-8167dfa7a78c","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6c901202-d912-4dfa-95db-fec2f7fc57e9","_cell_guid":"e8f6660a-1b0d-419f-bfff-8734e6e9842e","trusted":true},"cell_type":"code","source":"import keras.backend as K","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b229777f-d21b-4c48-92ec-c819f00da5e8","_cell_guid":"5dc30d63-1c9b-48b2-b87c-b1e513b7c8f9","trusted":true},"cell_type":"markdown","source":"### Loss Function - Quadratic Weighted Kappa\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"def quadratic_kappa_coefficient(y_true, y_pred):\n    y_true = K.cast(y_true, \"float32\")\n    n_classes = K.cast(y_pred.shape[-1], \"float32\")\n    weights = K.arange(0, n_classes, dtype=\"float32\") / (n_classes - 1)\n    weights = (weights - K.expand_dims(weights, -1)) ** 2\n\n    hist_true = K.sum(y_true, axis=0)\n    hist_pred = K.sum(y_pred, axis=0)\n\n    E = K.expand_dims(hist_true, axis=-1) * hist_pred\n    E = E / K.sum(E, keepdims=False)\n\n    O = K.transpose(K.transpose(y_true) @ y_pred)  # confusion matrix\n    O = O / K.sum(O)\n\n    num = weights * O\n    den = weights * E\n\n    QWK = (1 - K.sum(num) / K.sum(den))\n    return QWK\n\ndef quadratic_kappa_loss(scale=2.0):\n    def _quadratic_kappa_loss(y_true, y_pred):\n        QWK = quadratic_kappa_coefficient(y_true, y_pred)\n        loss = -K.log(K.sigmoid(scale * QWK))\n        return loss\n        \n    return _quadratic_kappa_loss","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"85ac5bad-c480-4879-96ac-aa2dc26cbbec","_cell_guid":"5c8c0c5f-4e09-49cb-afdd-a70c940db22f","trusted":true},"cell_type":"markdown","source":"### Build the model\n\nPerform transfer learning from VGG16 with imagenet weights. \nIn order to load the weights and use it without activating the internet, in the notebook, go to \"Data\" -> \"Add data\" and search \"keras\" then select \"Keras Pretrained Models\".\n\nFreeze the conv layers and only train the top layers."},{"metadata":{"_uuid":"e853dd4e-599f-4a16-afd9-be727929cd21","_cell_guid":"b77e2e82-4ffc-4b44-933d-0fc2a94c5d9f","trusted":true},"cell_type":"code","source":"from keras.applications.vgg16 import VGG16\nfrom keras import models, Model\nfrom keras.layers import Input,Dense, Dropout, Flatten, Conv2D, MaxPool2D\nfrom keras.optimizers import Adam\nfrom keras.losses import categorical_crossentropy","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9c51e71c-dd4f-40b5-ad69-62a91070308c","_cell_guid":"9e6bd177-92f6-4428-90c1-98b54a0d8d44","trusted":true},"cell_type":"code","source":"input_shape = (256, 256, 3)\n\nbase_net = VGG16(weights='../input/keras-pretrained-models/vgg16_weights_tf_dim_ordering_tf_kernels_notop.h5', include_top=False, input_shape=input_shape)\nfor layer in base_net.layers:\n    layer.trainable = False","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We choose a softmax activation on the last layer because our classes are mutually exclusive. We want the algorithm to choose only one class, the one with the highest probability, therefore the probabilities must sum up to 1. \n\n(in contrast, the sigmoid activation function will output independent probabilities and can be used when Eg. a patient might have multiple diseases - the output might be multiple classes)"},{"metadata":{"_uuid":"562c629f-5660-420c-87c5-58d7134e81c7","_cell_guid":"a2131017-ecc2-4163-87f8-7c9c613e39bf","trusted":true},"cell_type":"code","source":"model = models.Sequential()\nmodel.add(base_net)\n\nmodel.add(Flatten())\nmodel.add(Dense(256, activation = \"relu\"))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(6, activation = \"softmax\"))\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e4efb864-a503-424a-8ec7-d4e4eaa4b328","_cell_guid":"e9951896-390c-4fe9-8174-5e5b0e9a4c0c","trusted":true},"cell_type":"code","source":"model = Model(inputs = model.input, outputs = model.output)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"07faa61d-a0b3-4d04-9b1c-bec4417ca6bf","_cell_guid":"978a8931-25da-4846-8e51-d5de689e87fa","trusted":true},"cell_type":"code","source":"#loss = categorical_crossentropy,\nmodel.compile(optimizer = Adam(lr=1e-3), loss = quadratic_kappa_loss(scale=6.0), \\\n             metrics = ['accuracy',quadratic_kappa_coefficient])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0ba8e310-3254-4128-88a0-83510f6fa522","_cell_guid":"ddc06b8b-20d5-41d5-8b88-21a344e6f437","trusted":true},"cell_type":"markdown","source":"### Create a data generator\n\n1. Create a DF which contains the image path + the label of that image. We will not use the masks at all at this stage.\n2. Use the 3rd version o the data (smallest size array) to speed up the process\n3. Create a labels array (Y) and a data array (X)\n4. Split the data in train & validation (use validation to also test at this stage)"},{"metadata":{"_uuid":"9a5f639c-e1a5-4666-937a-2f83a9a7c4e1","_cell_guid":"fca970b2-94ef-48ec-b8f7-9fe172cb949a","trusted":true},"cell_type":"code","source":"from pathlib import Path\nimport pandas as pd\nimport numpy as np\nimport skimage.io\nimport cv2\n\nimport random\nfrom sklearn.model_selection import train_test_split\nfrom keras.callbacks.callbacks import ModelCheckpoint, EarlyStopping\n\nfrom sklearn.preprocessing import OneHotEncoder\n\nfrom matplotlib.pyplot import imshow","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"716644e7-ca53-4ce5-aa28-2c20075e7de0","_cell_guid":"b36e7d51-4cd0-4915-9323-b24ab6532009","trusted":true},"cell_type":"code","source":"HOME = Path(\"../input/prostate-cancer-grade-assessment\")\nTRAIN = Path(\"train_images\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"59760e3b-847f-468b-a4f2-f5ef0de4fbeb","_cell_guid":"77109b9f-4cbc-4f09-b0fb-55b0c7018656","trusted":true},"cell_type":"code","source":"train_ann = pd.read_csv(HOME/'train.csv')\ntrain_ann['image_path'] = [str(HOME/TRAIN/image_name) + \".tiff\" \\\n                           for image_name in train_ann['image_id']]\ntrain_ann.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c54b8af4-39e5-4713-a965-4db8c23c4426","_cell_guid":"47057354-cad1-4310-9dbd-d24b13955dc0","trusted":true},"cell_type":"markdown","source":"### Data Encoder for the labels\n\n... as we need them to be represented as dummy variables. Each response will be an array of length 6. Eg. class 3 will be represented as [0,0,0,1,0,0]"},{"metadata":{"_uuid":"661fa259-819e-4bf1-8a02-f4d520167d01","_cell_guid":"4ccd7983-7a39-45ef-b5fc-9bf8e1bc5dae","trusted":true},"cell_type":"code","source":"enc = OneHotEncoder(handle_unknown = 'ignore')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0f7aad79-44c6-4eea-a43f-9d59cf18f3d1","_cell_guid":"0a438b58-dace-4560-bff4-60e5239e68ee","trusted":true},"cell_type":"code","source":"enc_labels = pd.DataFrame(enc.fit_transform(train_ann[['isup_grade']]).toarray())\n\ntrain_ann = pd.merge(train_ann, enc_labels, left_index=True, right_index=True)\ntrain_ann.head(8)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1ac68005-c901-495d-b27f-b5ed22f4de8d","_cell_guid":"d5c56453-9fe8-443c-95a5-23ec329b4ba9","trusted":true},"cell_type":"markdown","source":"### Data Generator\n\n- First of all, take either train of val pandaDF. \n- Shuffle the rows and randomly select a number, equal to your batch size.\n- Read the selected images (from the path column) and resize them in get_image()\n- Output the data array as well as the labels corresponding to that data."},{"metadata":{"_uuid":"bdea433a-bd62-40cf-af4f-b04bea5365fb","_cell_guid":"5e0a36c5-03aa-49e3-9812-3b7bd99cb08c","trusted":true},"cell_type":"code","source":"# Function to get one image\n\ndef get_image(image_location):\n    image = skimage.io.MultiImage(image_location)\n    # take the smallest image size\n    image = image[-1]\n    # resize the image to the desired size\n    image = cv2.resize(image, (input_shape[0], input_shape[1]))\n    \n    return image","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0efeaea4-f15c-4e4d-b085-c5f5d9fc4b87","_cell_guid":"5100e56a-865e-4e52-9223-18a4bee75cb9","trusted":true},"cell_type":"code","source":"# Function that shuffles annotation rows and chooses batch_size samples\n#sequence = range(len(annotation_file))\n\ndef get_batch_ids(sequence, batch_size):\n    sequence = list(sequence)\n    random.shuffle(sequence)\n    batch = random.sample(sequence, batch_size)\n    return batch","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c4c42bb0-2e32-4ae6-993e-64e770cc73f4","_cell_guid":"22981c2b-a53a-473f-9172-6fe00d5d99fd","trusted":true},"cell_type":"code","source":"# Basic data generator -> Next: add augmentation = False\n\ndef data_generator(data, batch_size):\n    while True:\n        data = data.reset_index(drop=True)\n        indices = list(data.index)\n\n        batch_ids = get_batch_ids(indices, batch_size)\n        batch = data.iloc[batch_ids]['image_path']\n\n        X = [get_image(x) for x in batch]\n        Y = data[[0, 1, 2, 3, 4, 5]].values[batch_ids]\n\n        # Convert X and Y to arrays\n        X = np.array(X)\n        Y = np.array(Y)\n\n        yield X, Y\n\n# data: should be a pandas DF (train or val) obtained from train_test_split\n# batch_size: is the size of the number of images passed through the net in one step","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"43948f04-c6d8-4f18-b7f7-4a5ff54a8542","_cell_guid":"9c9db62c-4a01-486f-a620-7a01a34490dc","trusted":true},"cell_type":"markdown","source":"### Split the data set in train and validation"},{"metadata":{"_uuid":"5a96218f-7988-4a06-96f5-2ce229df3d8a","_cell_guid":"16159bb0-d821-4380-a1b9-9e46a4a63fa4","trusted":true},"cell_type":"code","source":"# Train -  Validation Split function\ntrain, val = train_test_split(train_ann, \\\n                              test_size = 0.1, \\\n                              random_state = 42)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"75eb5bec-d68a-4b58-9cb4-925b1eaf480c","_cell_guid":"513ffc08-8b2e-4ea1-aee2-37ca1d90d4a6","trusted":true},"cell_type":"code","source":"# Some checkpoints\nmodel_checkpoint = ModelCheckpoint('./model_01.h5', monitor = 'val_loss', verbose=0, save_best_only=True, save_weights_only=True)\nearly_stop = EarlyStopping(monitor='val_loss',patience=5,verbose=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"35896487-5cea-4d94-85e3-d165bf27c4cc","_cell_guid":"469ebeb5-415c-40bb-9431-2a7943a1986d","trusted":true},"cell_type":"markdown","source":"### Fit the model (Train)"},{"metadata":{"_uuid":"85abef76-6bcb-4dcf-905f-efc41c3ec9e1","_cell_guid":"3cebc633-bf58-4f10-b758-1c1f5357508d","trusted":true},"cell_type":"code","source":"EPOCHS = 30 \nBS = 100\n\nhistory = model.fit_generator(generator = data_generator(train, BS),\n                              validation_data = data_generator(val, BS),\n                              epochs = EPOCHS,\n                              verbose = 1,\n                              #steps_per_epoch = len(train)// BS,\\\n                              steps_per_epoch = 20,\n                              validation_steps = 20, \n                              #validation_steps = len(val)// BS,\\\n                              callbacks =[model_checkpoint, early_stop])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# summarize history for loss\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c7fe7fc2-28d6-4deb-b790-e45de03632d4","_cell_guid":"8084ec5a-2758-4fa2-b1b1-ccdb1747a412","trusted":true},"cell_type":"markdown","source":"## Predict on the Test Data \n- sample submission"},{"metadata":{"_uuid":"fd0270d2-855d-4871-96e4-5f23153eb184","_cell_guid":"64e7bf08-3bee-44ca-a115-daa147edb05e","trusted":true},"cell_type":"code","source":"initial_sample_submission = pd.read_csv('../input/prostate-cancer-grade-assessment/sample_submission.csv')\nTEST = Path(\"test_images\")\ntest_ann = pd.read_csv(HOME/'test.csv')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"14959fbe-9176-4da8-ac81-513dd3e85617","_cell_guid":"9f554be3-2387-4ad0-a685-921752efd40e","trusted":true},"cell_type":"code","source":"if os.path.exists(f'../input/prostate-cancer-grade-assessment/test_images/'):\n    print('inference!')\n\n    predictions = []\n    for img_id in test_ann['image_id']:\n        img = str(HOME/TEST/img_id) + \".tiff\"\n        print(img)\n        image = get_image(img)\n        image = image[np.newaxis,:]\n        prediction = model.predict(image)\n        # if we have 1 at multiple locations\n        ind = np.where(prediction == np.amax(prediction))\n        final_prediction = random.sample(list(ind[1]), 1)[0].astype(int)\n        predictions.append(final_prediction)\n\n    sample_submission = pd.DataFrame()\n    sample_submission['image_id'] = test_ann['image_id']\n    sample_submission['isup_grade'] = predictions\n    sample_submission\n\n    sample_submission.to_csv('submission.csv', index=False)\n    sample_submission.head()\nelse:\n    print('Test Images folder does not exist! Save the sample_submission.csv!')\n    initial_sample_submission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":4}