{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom tqdm.notebook import tqdm\nimport zipfile","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:13:45.127517Z","iopub.execute_input":"2023-09-20T21:13:45.127929Z","iopub.status.idle":"2023-09-20T21:13:45.256338Z","shell.execute_reply.started":"2023-09-20T21:13:45.127888Z","shell.execute_reply":"2023-09-20T21:13:45.255334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport skimage.io\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport PIL.Image\nfrom sklearn.model_selection import StratifiedKFold\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import cohen_kappa_score\nfrom tqdm import tqdm_notebook as tqdm","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:13:45.258055Z","iopub.execute_input":"2023-09-20T21:13:45.258383Z","iopub.status.idle":"2023-09-20T21:13:46.667215Z","shell.execute_reply.started":"2023-09-20T21:13:45.258352Z","shell.execute_reply":"2023-09-20T21:13:46.666202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '../input/prostate-cancer-grade-assessment'\ndf_train = pd.read_csv(os.path.join(data_dir, 'train.csv'))\n","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:13:46.668543Z","iopub.execute_input":"2023-09-20T21:13:46.669371Z","iopub.status.idle":"2023-09-20T21:13:46.705821Z","shell.execute_reply.started":"2023-09-20T21:13:46.669335Z","shell.execute_reply":"2023-09-20T21:13:46.704662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image ids in NAME variable with their respective classes in CLASS variable\nNAME=['0005f7aaab2800f6170c399693a96917','000920ad0b612851f8e01bcc880d9b3d','003046e27c8ead3e3db155780dc5498e','0032bfa835ce0f43a92ae0bbab6871cb','00bbc1482301d16de3ff63238cfd0b34',\n     '00d7ec94436e3a1416a3b302914957d3','0068d4c7529e34fd4c9da863ce01a161','006f6aa35a78965c92fffd1fbd53a058','0018ae58b01bdadc8e347995b69f99aa','001c62abd11fa4b57bf7a6c603a11bb9',\n     '00928370e2dfeb8a507667ef1d4efcbb','00c15b23b30a5ba061358d9641118904']\nCLASS=[0,0,1,1,2,2,3,3,4,4,5,5]\n","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:13:46.709821Z","iopub.execute_input":"2023-09-20T21:13:46.710156Z","iopub.status.idle":"2023-09-20T21:13:46.715635Z","shell.execute_reply.started":"2023-09-20T21:13:46.710128Z","shell.execute_reply":"2023-09-20T21:13:46.714693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=[]\ncount=0\nfor i in NAME:\n df.append(df_train[df_train['image_id']==i])\n count+=1\ndf_train=pd.DataFrame(np.array(df).reshape(12,4),columns=['image_id','data_provider','isup_grade','gleason_score'])\ndf_train","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:13:46.717518Z","iopub.execute_input":"2023-09-20T21:13:46.718327Z","iopub.status.idle":"2023-09-20T21:13:46.782243Z","shell.execute_reply.started":"2023-09-20T21:13:46.718291Z","shell.execute_reply":"2023-09-20T21:13:46.781079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN = '../input/prostate-cancer-grade-assessment/train_images/'\nMASKS = '../input/prostate-cancer-grade-assessment/train_label_masks/'\nOUT_TRAIN = 'train.zip'\nOUT_MASKS = 'masks.zip'\nsz = 256 # Size of each tile\nN = 36 # Total no of tiles","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:13:46.783878Z","iopub.execute_input":"2023-09-20T21:13:46.784339Z","iopub.status.idle":"2023-09-20T21:13:46.790057Z","shell.execute_reply.started":"2023-09-20T21:13:46.784303Z","shell.execute_reply":"2023-09-20T21:13:46.789074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tile(img, mask):\n    result = []\n    shape = img.shape\n    pad0,pad1 = (sz - shape[0]%sz)%sz, (sz - shape[1]%sz)%sz\n    img = np.pad(img,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],\n                constant_values=255)\n    mask = np.pad(mask,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],\n                constant_values=0)\n    img = img.reshape(img.shape[0]//sz,sz,img.shape[1]//sz,sz,3)\n    img = img.transpose(0,2,1,3,4).reshape(-1,sz,sz,3)\n    mask = mask.reshape(mask.shape[0]//sz,sz,mask.shape[1]//sz,sz,3)\n    mask = mask.transpose(0,2,1,3,4).reshape(-1,sz,sz,3)\n    if len(img) < N:\n        mask = np.pad(mask,[[0,N-len(img)],[0,0],[0,0],[0,0]],constant_values=0)\n        img = np.pad(img,[[0,N-len(img)],[0,0],[0,0],[0,0]],constant_values=255)\n    idxs = np.argsort(img.reshape(img.shape[0],-1).sum(-1))[:N]\n    img = img[idxs]\n    mask = mask[idxs]\n    for i in range(len(img)):\n        result.append({'img':img[i], 'mask':mask[i], 'idx':i})\n    return result","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:13:46.791502Z","iopub.execute_input":"2023-09-20T21:13:46.792523Z","iopub.status.idle":"2023-09-20T21:13:46.806807Z","shell.execute_reply.started":"2023-09-20T21:13:46.792466Z","shell.execute_reply":"2023-09-20T21:13:46.805594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_tot,x2_tot = [],[]\nnames = [name[:-10] for name in os.listdir(MASKS)]\nwith zipfile.ZipFile(OUT_TRAIN, 'w') as img_out,\\\n zipfile.ZipFile(OUT_MASKS, 'w') as mask_out:\n    for name in tqdm(NAME):\n        img = skimage.io.MultiImage(os.path.join(TRAIN,name+'.tiff'))[-1]\n        mask = skimage.io.MultiImage(os.path.join(MASKS,name+'_mask.tiff'))[-1]\n        tiles = tile(img,mask)\n        for t in tiles:\n            img,mask,idx = t['img'],t['mask'],t['idx']\n            x_tot.append((img/255.0).reshape(-1,3).mean(0))\n            x2_tot.append(((img/255.0)**2).reshape(-1,3).mean(0)) \n            #if read with PIL RGB turns into BGR\n            img = cv2.imencode('.png',cv2.cvtColor(img, cv2.COLOR_RGB2BGR))[1]\n            img_out.writestr(f'{name}_{idx}.png', img)\n            mask = cv2.imencode('.png',mask[:,:,0])[1]\n            mask_out.writestr(f'{name}_{idx}.png', mask)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:13:46.808213Z","iopub.execute_input":"2023-09-20T21:13:46.808632Z","iopub.status.idle":"2023-09-20T21:15:13.501051Z","shell.execute_reply.started":"2023-09-20T21:13:46.808598Z","shell.execute_reply":"2023-09-20T21:15:13.500042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! unzip '/kaggle/working/train.zip'","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:15:13.502778Z","iopub.execute_input":"2023-09-20T21:15:13.503164Z","iopub.status.idle":"2023-09-20T21:15:14.984894Z","shell.execute_reply.started":"2023-09-20T21:15:13.503129Z","shell.execute_reply":"2023-09-20T21:15:14.982975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loading all the tile images into img_train array with 12 rows\nimport cv2\nout_dir='/kaggle/working/'\nimg_train=np.empty((12,6*256,6*256,3))\ncount=-1\nfor i in NAME:\n   img_var=[]\n   count+=1\n   for j in range(0,36): \n     img_var.append(cv2.imread(out_dir+str(i)+'_'+str(j)+'.png'))\n   img_var=np.array(img_var)\n   img_var=img_var.reshape(6*256,6*256,3)\n   img_train[count]=img_var\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:15:14.993055Z","iopub.execute_input":"2023-09-20T21:15:14.993373Z","iopub.status.idle":"2023-09-20T21:15:15.996691Z","shell.execute_reply.started":"2023-09-20T21:15:14.993344Z","shell.execute_reply":"2023-09-20T21:15:15.995689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_train=img_train.reshape(12,-1) # for smote function to work, it requires only 2 dimensional array","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:15:15.998098Z","iopub.execute_input":"2023-09-20T21:15:15.998564Z","iopub.status.idle":"2023-09-20T21:15:16.004795Z","shell.execute_reply.started":"2023-09-20T21:15:15.998529Z","shell.execute_reply":"2023-09-20T21:15:16.003494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 12 to 24 images using smote analysis\nfrom collections import Counter\nfrom imblearn.over_sampling import SMOTE\ncounter = Counter(CLASS)\nprint('Before',counter)\n# oversampling the train dataset using SMOTE\nsmt = SMOTE(sampling_strategy={0:4,1:4,2:4,3:4,4:4,5:4},k_neighbors=1)\n#X_train, y_train = smt.fit_resample(X_train, y_train)\nX_train_sm, y_train_sm = smt.fit_resample(img_train, CLASS)\n\ncounter = Counter(y_train_sm)\nprint('After',counter)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:15:16.006106Z","iopub.execute_input":"2023-09-20T21:15:16.007044Z","iopub.status.idle":"2023-09-20T21:15:19.88335Z","shell.execute_reply.started":"2023-09-20T21:15:16.006981Z","shell.execute_reply":"2023-09-20T21:15:19.882428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train=X_train_sm.reshape(24,6*256,6*256,3) # reshaping array after completion of smote analysis","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:15:19.884764Z","iopub.execute_input":"2023-09-20T21:15:19.889676Z","iopub.status.idle":"2023-09-20T21:15:19.89573Z","shell.execute_reply.started":"2023-09-20T21:15:19.889639Z","shell.execute_reply":"2023-09-20T21:15:19.894657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_train=pd.get_dummies(y_train_sm) # onehotencoding","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:15:19.897301Z","iopub.execute_input":"2023-09-20T21:15:19.89793Z","iopub.status.idle":"2023-09-20T21:15:19.918672Z","shell.execute_reply.started":"2023-09-20T21:15:19.897897Z","shell.execute_reply":"2023-09-20T21:15:19.917769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining the model\nimport tensorflow.keras as K\nfrom tensorflow.keras.layers import Input\ninput_t=Input(shape=(6*256,6*256,3))\nres_model=K.applications.ResNet50(include_top=False,weights=\"imagenet\",input_tensor=input_t)\nmodel = K.models.Sequential()\nmodel.add(res_model)\nmodel.add(K.layers.Flatten())\nmodel.add(K.layers.Dense(6, activation='softmax'))","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:15:19.920287Z","iopub.execute_input":"2023-09-20T21:15:19.921114Z","iopub.status.idle":"2023-09-20T21:15:34.22119Z","shell.execute_reply.started":"2023-09-20T21:15:19.921076Z","shell.execute_reply":"2023-09-20T21:15:34.220177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='categorical_crossentropy',optimizer=K.optimizers.RMSprop(lr=2e-5),metrics=['accuracy'])\nhistory=model.fit(X_train,Y_train,batch_size=1,epochs=10)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T21:15:34.222543Z","iopub.execute_input":"2023-09-20T21:15:34.222889Z","iopub.status.idle":"2023-09-20T21:18:08.44829Z","shell.execute_reply.started":"2023-09-20T21:15:34.222857Z","shell.execute_reply":"2023-09-20T21:18:08.447282Z"},"trusted":true},"execution_count":null,"outputs":[]}]}