{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import sys\nprint(sys.version)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install -U efficientnet","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport matplotlib.pyplot as plt\nfrom skimage.io import MultiImage,imsave,imread\nfrom skimage.transform import resize,rescale\nfrom skimage.color import rgb2gray\nfrom keras.layers import Input,Cropping2D,GlobalAveragePooling2D,Concatenate,Dense,Conv2D\nfrom keras.models import Model,load_model\nimport keras.applications as kl\nfrom keras.backend import name_scope\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score\nfrom tqdm import tqdm\nimport tensorflow as tf\nfrom keras.utils import Sequence\nfrom keras.optimizers import Adam,Adamax\nfrom sklearn.utils import shuffle,class_weight\nfrom keras.utils import to_categorical\nimport efficientnet.keras as efn\n\nfrom albumentations import (\n    HorizontalFlip, IAAPerspective, ShiftScaleRotate, CLAHE, RandomRotate90,\n    Transpose, ShiftScaleRotate, Blur, OpticalDistortion, GridDistortion, HueSaturationValue,\n    IAAAdditiveGaussianNoise, GaussNoise, MotionBlur, MedianBlur, RandomBrightnessContrast, IAAPiecewiseAffine,\n    IAASharpen, IAAEmboss, Flip, OneOf, Compose\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import keras\nimport skimage","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('skimage version: ',skimage.__version__)\nprint('keras version:',keras.__version__)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"main_path='../input/prostate-cancer-grade-assessment/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_dim=(224,224,3)\nBATCH_SIZE=16\nEPOCHS=10","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_df=pd.read_csv(os.path.join(main_path,'train.csv'))\ntest_df=pd.read_csv(os.path.join(main_path,'test.csv'))\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.gleason_score.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mapper = {'3+3':1,'0+0':0,'3+4':2,'4+3':3,'4+4':4,'negative':0,'4+5':5,'5+4':6,'5+5':7,'3+5':8,'5+3':9}\ntrain_df['gleason_mapper'] = train_df.gleason_score.map(mapper)\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.loc[train_df.gleason_score=='negative']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Lets take a look at isup_grade(target feature)\ntrain_df['gleason_mapper'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Plot some slides\nrows,cols=3,4\nfig=plt.figure(figsize=(10,10))\nfor i in range(1,rows*cols+1):\n    img=MultiImage(os.path.join(main_path,'train_images',train_df.loc[i-1,'image_id']+'.tiff'))\n    img = resize(img[-1], (512, 512))\n    fig.add_subplot(rows,cols,i)\n    plt.imshow(img)\n    plt.title('gleason_mapper: '+str(train_df.loc[i-1,'gleason_mapper']))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Convert and save images**"},{"metadata":{"trusted":true},"cell_type":"code","source":"#for id_ in tqdm(train_df['image_id']):\n#    img=MultiImage(os.path.join(main_path,'train_images',id_+'.tiff'))\n#    img = resize(img[-1], (1024, 1024))\n#    imsave(id_+'.jpg',img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"colums=['image_id','gleason_mapper']\ntrain_df,val_df=train_test_split(train_df[colums],test_size=0.15)\nprint('Train shape: {}'.format(train_df.shape))\nprint('Validation shape: {}'.format(val_df.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sorted(train_df.gleason_mapper.unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sorted(val_df.gleason_mapper.unique())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Generator**"},{"metadata":{"trusted":true},"cell_type":"code","source":"class Generator(Sequence):\n    def __init__(self,input_data,batch_size=BATCH_SIZE,dims=image_dim,is_shuffle=True,n_classes=10,is_train=True):\n        self.image_ids=input_data[0]\n        self.labels=input_data[1]\n        self.batch_size=batch_size\n        self.dims=image_dim\n        self.shuffle=is_shuffle\n        self.n_classes=n_classes\n        self.is_train=is_train\n        self.on_epoch_end()\n    \n    def __len__(self):\n        return int(np.floor(len(self.image_ids) / self.batch_size))\n    \n    def on_epoch_end(self):\n        self.indexes = np.arange(len(self.image_ids))\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n    \n    def __getitem__(self, index):\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n\n        image_ids_temp = [self.image_ids[k] for k in indexes]\n        labels_temp = [self.labels[k] for k in indexes]\n\n        # Generate data\n        X, y = self.__data_generation(image_ids_temp,labels_temp)\n\n        return X, y\n    \n    def augment_flips_color(self,p=.5):\n        return Compose([\n            Flip(),\n            RandomRotate90(),\n            Transpose(),\n            HorizontalFlip(),\n            ShiftScaleRotate(shift_limit=0.0625, scale_limit=0.50, rotate_limit=45, p=.75),\n            Blur(blur_limit=3),\n        ], p=p)\n    \n    def __data_generation(self, list_IDs_temp,lbls):\n        X = np.zeros((self.batch_size, *self.dims))\n        y = np.zeros((self.batch_size), dtype=int)\n\n        # Generate data\n        for i, ID in enumerate(list_IDs_temp):\n            # Store sample\n            img=MultiImage(os.path.join(main_path,'train_images',ID+'.tiff'))\n            img = resize(img[-1], (self.dims[0], self.dims[1]))\n            #Augmentation\n            if self.is_train:\n                aug = self.augment_flips_color(p=1)\n                img = aug(image=img)['image']\n                \n            X[i] = img\n\n            # Store class\n            y[i] = lbls[i]\n\n        return X, to_categorical(y, num_classes=self.n_classes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_gen=Generator([train_df['image_id'].values,train_df['gleason_mapper'].values])\nval_gen=Generator([val_df['image_id'].values,val_df['gleason_mapper'].values],is_shuffle=False,is_train=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.callbacks import Callback\nclass QWKEvaluation(Callback):\n    def __init__(self, validation_data=(), batch_size=BATCH_SIZE, interval=1):\n        super(Callback, self).__init__()\n\n        self.interval = interval\n        self.batch_size = batch_size\n        self.valid_generator, self.y_val = validation_data\n        self.history = []\n\n    def on_epoch_end(self, epoch, logs={}):\n        if epoch % self.interval == 0:\n            y_pred = self.model.predict_generator(generator=self.valid_generator,\n                                                  steps=np.ceil(float(len(self.y_val)) / float(self.batch_size)),\n                                                  workers=1, use_multiprocessing=False,\n                                                  verbose=1)\n            def flatten(y):\n                return np.argmax(y, axis=1).reshape(-1)\n            \n            score = cohen_kappa_score(self.y_val,\n                                      flatten(y_pred),\n                                      labels=[0,1,2,3,4,5,6,7,8,9],\n                                      weights='quadratic')\n            print(\"\\n epoch: %d - QWK_score: %.6f \\n\" % (epoch+1, score))\n            self.history.append(score)\n            if score >= max(self.history):\n                print('saving checkpoint: ', score)\n                self.model.save('classifier.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_weights = class_weight.compute_class_weight('balanced',\n                                                 np.unique(train_df['gleason_mapper']),\n                                                   train_df['gleason_mapper'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"qwk = QWKEvaluation(validation_data=(val_gen, np.asarray(val_df['gleason_mapper'][:val_gen.__len__()*BATCH_SIZE])),\n                    batch_size=BATCH_SIZE, interval=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Model     \ninp=Input(shape=image_dim)\n\nbase_model=base_model=efn.EfficientNetB6(weights='imagenet',include_top=False,input_tensor=inp)\n\nfor layer in base_model.layers:\n    layer.trainable=True\n\nfeat=GlobalAveragePooling2D()(base_model.output)\nout=Dense(10,activation='softmax')(feat)\nmodel=Model(inp,out)\nmodel.compile(loss='binary_crossentropy',optimizer=Adam(0.001),metrics=['acc'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history=model.fit_generator(train_gen,epochs=EPOCHS,steps_per_epoch=100,validation_data=val_gen,\n                    validation_steps=100,\n                    callbacks=[qwk],class_weight=class_weights)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"loss = history.history['loss']\nval_loss = history.history['val_loss']\nscore=qwk.history\nepochs=range(1,len(loss)+1)\nplt.plot(epochs,loss,'b',color='red',label='Training Loss')\nplt.plot(epochs,val_loss,'b',color='blue',label='Validation Loss')\nplt.title('Training and Validation Loss')\nplt.legend()\nplt.figure()\nplt.plot(epochs,score,'b',color='red',label='Validation Kappa')\nplt.legend()\nplt.figure()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del train_gen,val_gen,train_df,val_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_df","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Test Prediction**"},{"metadata":{"trusted":true},"cell_type":"code","source":"test_dir='../input/prostate-cancer-grade-assessment/test_images'\nif os.path.exists(test_dir):\n    model=load_model('classifier.h5')\n    predicted=[]\n    for ID in test_df['image_Id']:\n        img=MultiImage(os.path.join(test_dir,ID+'.tiff'))\n        img = resize(img[-1], (image_dim[0], image_dim[1]))\n        preds=model.predict(np.expand_dims(img,0))\n        preds = np.argmax(preds,axis=0)\n        predicted.append(preds)\n        \n    submission=pd.DataFrame({'image_id':test_df['image_id'],'isup_grade':predicted})\n    submission.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model=load_model('classifier.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}