{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\nimport numpy as np \nimport pandas as pd \n\nimport cv2\nimport tifffile as tifi\nfrom tqdm.notebook import tqdm\nimport matplotlib.pylab as plt\n\n# from sklearn.ensemble import RandomForestClassifier\n\nimport tensorflow as tf\nfrom scipy.special import softmax\n\nimport pickle\nimport os\nimport warnings\nwarnings.filterwarnings(\"ignore\")\ngc.enable()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-03T23:33:50.693996Z","iopub.execute_input":"2022-10-03T23:33:50.694478Z","iopub.status.idle":"2022-10-03T23:33:56.613345Z","shell.execute_reply.started":"2022-10-03T23:33:50.694383Z","shell.execute_reply":"2022-10-03T23:33:56.611892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_path = '../input/new-ds6-classifier-mayo-strip-ai/'\n\ninput_path = '../input/mayo-clinic-strip-ai/'\nimage_path = '../input/mayo-clinic-strip-ai/test/'\n\ndf_test = pd.read_csv(input_path+'test.csv', header=0)\ndf_sub = pd.read_csv(input_path+'sample_submission.csv', header=0)\n\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-03T23:33:56.615415Z","iopub.execute_input":"2022-10-03T23:33:56.616182Z","iopub.status.idle":"2022-10-03T23:33:56.658496Z","shell.execute_reply.started":"2022-10-03T23:33:56.616133Z","shell.execute_reply":"2022-10-03T23:33:56.657281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TILE_SIZE = 224\nDOWN_SMPL = 6\n\nFILE_SIZE = 2.2  ## in Gb","metadata":{"execution":{"iopub.status.busy":"2022-10-03T23:33:56.660155Z","iopub.execute_input":"2022-10-03T23:33:56.660489Z","iopub.status.idle":"2022-10-03T23:33:56.665096Z","shell.execute_reply.started":"2022-10-03T23:33:56.660456Z","shell.execute_reply":"2022-10-03T23:33:56.664064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_tif(path):\n    image = tifi.imread(path)\n    shape = image.shape\n    image_resize = cv2.resize(image, (shape[1]//DOWN_SMPL,shape[0]//DOWN_SMPL), \n                              interpolation=cv2.INTER_LINEAR)\n    del image\n    gc.collect()\n    \n    return image_resize\n\n\nDS_list = []\n# cn = 0\nfor i,fname in enumerate(tqdm(df_test['image_id'])):\n    path = image_path+fname+'.tif'\n    size = os.path.getsize(path)\n    if size/1e9 >= 1 and size/1e9 <= FILE_SIZE:\n\n        im_resize = read_tif(path)\n        cv2.imwrite('./'+fname+'.png', im_resize)\n        DS_list.append(fname)\n        gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-03T23:33:56.666657Z","iopub.execute_input":"2022-10-03T23:33:56.666977Z","iopub.status.idle":"2022-10-03T23:34:39.237403Z","shell.execute_reply.started":"2022-10-03T23:33:56.666947Z","shell.execute_reply":"2022-10-03T23:34:39.236245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_ext = tf.keras.models.load_model(model_path)\nwt = np.load(model_path+'pred_weights.npy', allow_pickle=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T23:34:39.240282Z","iopub.execute_input":"2022-10-03T23:34:39.240657Z","iopub.status.idle":"2022-10-03T23:34:54.391027Z","shell.execute_reply.started":"2022-10-03T23:34:39.240623Z","shell.execute_reply":"2022-10-03T23:34:54.389858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pipeline(path, feat_ext, wt, DS=True, sz=TILE_SIZE,  N=25):\n    \n    if DS == True: \n        image = tifi.imread(path)\n        shape = image.shape\n        img = cv2.resize(image, (shape[1]//DOWN_SMPL,shape[0]//DOWN_SMPL), interpolation=cv2.INTER_LINEAR)\n        del image\n        gc.collect()\n    elif DS == False:\n        img = cv2.imread(path)\n    \n    shape = img.shape\n    pad0,pad1 = (sz - shape[0]%sz)%sz, (sz - shape[1]%sz)%sz\n    img = np.pad(img,[[pad0//2,pad0-pad0//2],[pad1//2,pad1-pad1//2],[0,0]],constant_values=255)\n    img = img.reshape(img.shape[0]//sz,sz,img.shape[1]//sz,sz,3)\n    img = img.transpose(0,2,1,3,4).reshape(-1,sz,sz,3)\n    \n    if len(img) < N:\n        img = np.pad(img,[[0,N-len(img)],[0,0],[0,0],[0,0]],constant_values=255)\n    idxs = np.argsort(img.reshape(img.shape[0],-1).sum(-1))[:N]\n    img = img[idxs]\n    \n    fsize = feat_ext.layers[-1].get_output_at(0).get_shape().as_list()[1]\n    features = np.zeros((len(img),fsize))\n    preds = np.zeros((len(img),2))\n    \n    for j,tile in enumerate(img):\n        features[j] = feat_ext.predict(tile.reshape(-1,TILE_SIZE,TILE_SIZE,3))\n        preds[j] = tf.keras.activations.sigmoid(np.dot(features[j],wt[0]) + wt[1]).numpy()   \n    \n    laa_max = max(preds[:,0])\n    ce_max = max(preds[:,1])\n    \n    return features, laa_max, ce_max","metadata":{"execution":{"iopub.status.busy":"2022-10-03T23:34:54.392767Z","iopub.execute_input":"2022-10-03T23:34:54.393114Z","iopub.status.idle":"2022-10-03T23:34:54.407896Z","shell.execute_reply.started":"2022-10-03T23:34:54.393081Z","shell.execute_reply":"2022-10-03T23:34:54.406335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def aggregator(features):\n    \n    ind_max = np.argmax(features[:])//features.shape[1]\n    \n    ## -- attention score for each feature vector wrt critical feature vector\n    atten = np.zeros(features.shape[0])\n    fc = features[ind_max]\n    for i,f in enumerate(features):\n        atten[i] = np.dot(fc,f)\n    \n    atten = softmax(atten/atten.max())\n    soft_feat = features.reshape(-1,features.shape[0])*atten\n    agg_features = np.sum(soft_feat, axis=1)\n    \n    \n    return agg_features","metadata":{"execution":{"iopub.status.busy":"2022-10-03T23:34:54.409401Z","iopub.execute_input":"2022-10-03T23:34:54.410105Z","iopub.status.idle":"2022-10-03T23:34:54.42795Z","shell.execute_reply.started":"2022-10-03T23:34:54.410069Z","shell.execute_reply":"2022-10-03T23:34:54.426467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cls_model = pickle.load(open(model_path+'rndfor_new4.pickle.dat', \"rb\"))","metadata":{"execution":{"iopub.status.busy":"2022-10-03T23:34:54.429405Z","iopub.execute_input":"2022-10-03T23:34:54.430494Z","iopub.status.idle":"2022-10-03T23:34:55.040736Z","shell.execute_reply.started":"2022-10-03T23:34:54.430458Z","shell.execute_reply":"2022-10-03T23:34:55.039299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(data=None, columns=df_test.columns)\n\nfor i,filename in enumerate(tqdm(df_test['image_id'])):\n    if filename in DS_list:\n        path = './'+filename+'.png'\n        DS = False\n    else:\n        path = image_path+filename+'.tif'\n        DS = True\n        \n    dat = df_test[df_test['image_id'] == filename]\n    \n    try:\n        size = os.path.getsize(path)\n    except:\n        size = FILE_SIZE*1.5*1e9\n\n    if (size > FILE_SIZE*1e9):\n        \n        dat['CE'] = 0.5\n        dat['LAA'] = 0.5\n\n    else:\n        try:\n            features, laa_max, ce_max = pipeline(path, feat_ext, wt, DS=DS)\n\n            dat['CE'] = 0.5\n            dat['LAA'] = 0.5\n            if len(features) > 2:\n\n                agg_feat = aggregator(features)\n#                 agg_feat = np.insert(agg_feat, 0, [laa_max, ce_max], axis=0)\n                pred = cls_model.predict_proba(agg_feat.reshape(1,-1))\n                dat['CE'] = pred[0][1]\n                dat['LAA'] = pred[0][0]\n\n        except:\n            dat['CE'] = 0.5\n            dat['LAA'] = 0.5\n            \n    df = pd.concat([df, dat], axis=0, ignore_index=True)\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-03T23:34:55.042227Z","iopub.execute_input":"2022-10-03T23:34:55.042761Z","iopub.status.idle":"2022-10-03T23:35:58.431062Z","shell.execute_reply.started":"2022-10-03T23:34:55.042704Z","shell.execute_reply":"2022-10-03T23:35:58.429935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_mean = df[['patient_id', 'CE', 'LAA']].groupby('patient_id').mean()\n\ndf_sub['CE'] = df_mean['CE'].to_list()\ndf_sub['LAA'] = df_mean['LAA'].to_list()\n\ndf_sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-03T23:35:58.4328Z","iopub.execute_input":"2022-10-03T23:35:58.433127Z","iopub.status.idle":"2022-10-03T23:35:58.453542Z","shell.execute_reply.started":"2022-10-03T23:35:58.433097Z","shell.execute_reply":"2022-10-03T23:35:58.452708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T23:35:58.45498Z","iopub.execute_input":"2022-10-03T23:35:58.455345Z","iopub.status.idle":"2022-10-03T23:35:58.481862Z","shell.execute_reply.started":"2022-10-03T23:35:58.455312Z","shell.execute_reply":"2022-10-03T23:35:58.480642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}