{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport skimage.io\n# Data visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport PIL\nfrom IPython.display import Image, display\nfrom plotly import graph_objs as go\nimport plotly.express as px\nimport plotly.figure_factory as ff\nimport openslide\n# Thư viện train model\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, Dataset\nfrom efficientnet_pytorch import model as enet\nfrom tqdm import tqdm_notebook as tqdm","metadata":{"execution":{"iopub.status.busy":"2022-03-16T01:02:03.954918Z","iopub.execute_input":"2022-03-16T01:02:03.95565Z","iopub.status.idle":"2022-03-16T01:02:03.961578Z","shell.execute_reply.started":"2022-03-16T01:02:03.955597Z","shell.execute_reply":"2022-03-16T01:02:03.96106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n!pip install efficientnet_pytorch","metadata":{"execution":{"iopub.status.busy":"2022-03-16T01:00:46.165458Z","iopub.execute_input":"2022-03-16T01:00:46.166495Z","iopub.status.idle":"2022-03-16T01:00:56.507308Z","shell.execute_reply.started":"2022-03-16T01:00:46.166451Z","shell.execute_reply":"2022-03-16T01:00:56.506383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_FOLDER = \"/kaggle/input/prostate-cancer-grade-assessment/\"\n!ls {BASE_FOLDER}","metadata":{"execution":{"iopub.status.busy":"2022-03-16T01:06:38.55494Z","iopub.execute_input":"2022-03-16T01:06:38.555528Z","iopub.status.idle":"2022-03-16T01:06:39.277549Z","shell.execute_reply.started":"2022-03-16T01:06:38.555494Z","shell.execute_reply":"2022-03-16T01:06:39.276739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Đọc các file trong data\ntrain = pd.read_csv(BASE_FOLDER + \"train.csv\")\ntest = pd.read_csv(BASE_FOLDER + \"test.csv\")\nsub = pd.read_csv(BASE_FOLDER + \"sample_submission.csv\")\nprint(train.shape)\n# Xem dữ liệu\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-16T01:06:46.49206Z","iopub.execute_input":"2022-03-16T01:06:46.492534Z","iopub.status.idle":"2022-03-16T01:06:46.551282Z","shell.execute_reply.started":"2022-03-16T01:06:46.492502Z","shell.execute_reply":"2022-03-16T01:06:46.550553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tìm các giá trị unique trong từng trường dữ liệu\nprint(\"unique ids: \", len(train.image_id.unique()))\nprint(\"unique data_provider: \", len(train.data_provider.unique()))\nprint(\"isup_grade: \", len(train.isup_grade.unique()))\nprint(\"gleasion_score: \", len(train.gleason_score.unique()))","metadata":{"execution":{"iopub.status.busy":"2022-03-16T01:08:03.276871Z","iopub.execute_input":"2022-03-16T01:08:03.277565Z","iopub.status.idle":"2022-03-16T01:08:03.295956Z","shell.execute_reply.started":"2022-03-16T01:08:03.277521Z","shell.execute_reply":"2022-03-16T01:08:03.295106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['gleason_score'].unique()\n# kiểm tra xem gleason_score 0+0 và điểm isup_grade tương ứng\nprint(train[train['gleason_score']=='0+0']['isup_grade'].unique())\nprint(train[train['gleason_score']=='negative']['isup_grade'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-03-16T01:10:09.505453Z","iopub.execute_input":"2022-03-16T01:10:09.505785Z","iopub.status.idle":"2022-03-16T01:10:09.523413Z","shell.execute_reply.started":"2022-03-16T01:10:09.505751Z","shell.execute_reply":"2022-03-16T01:10:09.52265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Do 'gleason scrore'  0+0 và negative đều có isup_grade = 0 => ta sẽ gộp 2 nhãn này vào 1.","metadata":{}},{"cell_type":"code","source":"train['gleason_score'] = train['gleason_score'].apply(lambda x: \"0+0\" if x==\"negative\" else x)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Do cá nhãn gleason-score 3+4 và 4+3 , 3+5, 5+3, 4+5, 5+4 đều map đến một điểm ISUP chung\n# Ta sẽ kiểm tra tính đúng đắn của nó\nprint(train[(train['gleason_score']=='3+4') | (train['gleason_score']=='4+3')]['isup_grade'].unique())\nprint(train[(train['gleason_score']=='3+5') | (train['gleason_score']=='5+3')]['isup_grade'].unique())\nprint(train[(train['gleason_score']=='5+4') | (train['gleason_score']=='4+5')]['isup_grade'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-03-16T02:04:15.956873Z","iopub.execute_input":"2022-03-16T02:04:15.958139Z","iopub.status.idle":"2022-03-16T02:04:15.993782Z","shell.execute_reply.started":"2022-03-16T02:04:15.958101Z","shell.execute_reply":"2022-03-16T02:04:15.992924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Nhận thấy gleson-soore 3+4 và 4+3 có 2 giá trị unique. Nên ta sẽ kiểm tra lại chúng","metadata":{}},{"cell_type":"code","source":"\n\nprint(train[train['gleason_score']=='3+4']['isup_grade'].unique())\nprint(train[train['gleason_score']=='4+3']['isup_grade'].unique())\n","metadata":{"execution":{"iopub.status.busy":"2022-03-16T02:08:16.589617Z","iopub.execute_input":"2022-03-16T02:08:16.589913Z","iopub.status.idle":"2022-03-16T02:08:16.601778Z","shell.execute_reply.started":"2022-03-16T02:08:16.58988Z","shell.execute_reply":"2022-03-16T02:08:16.601166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Điểm gleson 3+4 có giá trị ISUP đúng, trong khi gleson 4+3 đang có dữ liệu bị nhầm lẫn thành ISUP = 2, => Tìm và loại bỏ dữ liệu bị đánh nhãn sai.","metadata":{}},{"cell_type":"code","source":"train[(train['isup_grade'] == 2) & (train['gleason_score'] == '4+3')]","metadata":{"execution":{"iopub.status.busy":"2022-03-16T02:14:01.478906Z","iopub.execute_input":"2022-03-16T02:14:01.480178Z","iopub.status.idle":"2022-03-16T02:14:01.499176Z","shell.execute_reply.started":"2022-03-16T02:14:01.480122Z","shell.execute_reply":"2022-03-16T02:14:01.498433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = train.groupby('isup_grade').count()['image_id'].reset_index().sort_values(by='image_id',ascending=False)\ntemp.style.background_gradient(cmap='Purples')","metadata":{"execution":{"iopub.status.busy":"2022-03-16T02:30:28.317694Z","iopub.execute_input":"2022-03-16T02:30:28.318541Z","iopub.status.idle":"2022-03-16T02:30:28.418643Z","shell.execute_reply.started":"2022-03-16T02:30:28.318498Z","shell.execute_reply":"2022-03-16T02:30:28.41796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop([7273],inplace=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train[train['gleason_score']=='0+0']['isup_grade']))\nprint(len(train[train['gleason_score']=='negative']['isup_grade']))","metadata":{"execution":{"iopub.status.busy":"2022-02-15T15:38:32.762661Z","iopub.execute_input":"2022-02-15T15:38:32.763447Z","iopub.status.idle":"2022-02-15T15:38:32.773692Z","shell.execute_reply.started":"2022-02-15T15:38:32.76341Z","shell.execute_reply":"2022-02-15T15:38:32.772838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = openslide.OpenSlide(os.path.join(BASE_FOLDER+\"train_images\", '005e66f06bce9c2e49142536caf2f6ee.tiff'))\npatch = example.read_region((17800,19500), 0, (256,256))\ndisplay(patch)\nprint(example.dimensions)\nplt.imshow(example.get_thumbnail(size=(600,400)))","metadata":{"execution":{"iopub.status.busy":"2022-02-16T15:26:22.582778Z","iopub.execute_input":"2022-02-16T15:26:22.583089Z","iopub.status.idle":"2022-02-16T15:26:23.409729Z","shell.execute_reply.started":"2022-02-16T15:26:22.583056Z","shell.execute_reply":"2022-02-16T15:26:23.408414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_values(image, max_size=(600,400)):\n    slide = openslide.OpenSlide(os.path.join(BASE_FOLDER+\"train_images\", f'{image}.tiff'))\n    # Tính toán kích thước của  pixel\n    f,ax = plt.subplots(2, figsize=(6,16))\n    #do openslide đo kích thước dưới đơn vị cm nên đổi ra micromet (chia cho 10000)\n    spacing = 1 / (float(slide.properties['tiff.XResolution'])/10000)\n    path = slide.read_region((1780,1950), 0 , (256,256))\n    # Hiện thị hình ành sử dụng thư viện matplotlib (2 hàng)\n    ax[0].imshow(path) # ảnh zoom với kích thước 256*256\n    ax[0].set_title('Zoomed Image')\n    \n    ax[1].imshow(slide.get_thumbnail(size=max_size))\n    ax[1].set_title('Full Image')\n    \n    # Hiện thị các thông số của ảnh\n    print(f\"File id: {slide}\")\n    print(f\"Dimensions:  {slide.dimensions}\")\n    print(f\"Microns per pixel: {spacing: .3f}\")\n    print(f\"Dowsample factor per level: {slide.level_downsamples}\")\n    print(f\"Dimensions of levels: {slide.level_dimensions}\\n\\n\")\n    \n   # print(f\"ISUP grade: {train.loc[image, 'isup_grade']}\")\n    #print(f\"Gleason score: {train.loc[image, 'gleason_score']}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-10T12:52:10.227907Z","iopub.execute_input":"2022-02-10T12:52:10.228198Z","iopub.status.idle":"2022-02-10T12:52:10.237073Z","shell.execute_reply.started":"2022-02-10T12:52:10.228167Z","shell.execute_reply":"2022-02-10T12:52:10.236197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_values('005e66f06bce9c2e49142536caf2f6ee')","metadata":{"execution":{"iopub.status.busy":"2022-02-10T12:52:15.045824Z","iopub.execute_input":"2022-02-10T12:52:15.046605Z","iopub.status.idle":"2022-02-10T12:52:15.789043Z","shell.execute_reply.started":"2022-02-10T12:52:15.046565Z","shell.execute_reply":"2022-02-10T12:52:15.788223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_folder = os.path.join(BASE_FOLDER, 'test_images')\nis_test = os.path.exists(image_folder)  # IF test_images is not exists, we will use some train images.\nimage_folder = image_folder if is_test else os.path.join(BASE_FOLDER, 'train_images')\nprint(image_folder)\ndf = test if is_test else train.loc[:100]\nprint(df.shape)\nfold = 0\n# Chia ảnh thành 36 ảnh kích thước 256*256\ntile_size = 256\nimage_size = 256\nn_tiles = 81\nbatch_size = 8\nnumm_workers = 4\ndevice = torch.device('cuda')","metadata":{"execution":{"iopub.status.busy":"2022-03-02T02:44:44.906545Z","iopub.execute_input":"2022-03-02T02:44:44.907532Z","iopub.status.idle":"2022-03-02T02:44:44.914632Z","shell.execute_reply.started":"2022-03-02T02:44:44.907478Z","shell.execute_reply":"2022-03-02T02:44:44.913863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tiến hành thực hiện Cross-Entropy","metadata":{}},{"cell_type":"code","source":"skf = StratifiedKFold(5, shuffle=True, random_state=42)\ntrain['fold'] = -1\nfor i, (train_idx, valid_idx) in enumerate(skf.split(train, train['isup_grade'])):\n    train.loc[valid_idx, 'fold'] = i\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-01T13:31:40.927707Z","iopub.execute_input":"2022-03-01T13:31:40.928216Z","iopub.status.idle":"2022-03-01T13:31:40.950261Z","shell.execute_reply.started":"2022-03-01T13:31:40.928181Z","shell.execute_reply":"2022-03-01T13:31:40.949541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Phần xử lý dữ liệu","metadata":{}},{"cell_type":"code","source":"\n\ndef get_tiles(img, mode=0):\n        result = []\n        h, w, c = img.shape    # Lấy ra kích thước của ảnh\n        # Chia ảnh cho tỷ lệ để thêm padding đối với những ảnh có kích thước ko chia hết cho 256(độ rộng ảnh mong muốn)\n        pad_h = (tile_size - h % tile_size) % tile_size + ((tile_size * mode) // 2)\n        pad_w = (tile_size - w % tile_size) % tile_size + ((tile_size * mode) // 2)\n        # padding trên dưới và trái phải của ảnh (padding bằng giá trị 255 thưc là bổ sung phần trắng vào cho ảnh)\n        img2 = np.pad(img,[[pad_h // 2, pad_h - pad_h // 2], [pad_w // 2,pad_w - pad_w//2], [0,0]], constant_values=255)\n        # reshape kích thước ảnh .tiff ban đầu\n        img3 = img2.reshape(\n            img2.shape[0] // tile_size,\n            tile_size,\n            img2.shape[1] // tile_size,\n            tile_size,\n            3\n        )\n\n        img3 = img3.transpose(0,2,1,3,4).reshape(-1, tile_size, tile_size,3)\n        # Tính tổng các pixel của từng ảnh sau khi đã chia kích thước (256*256*3) nếu tổng này < (tile_size**2*3*255) tức là\n        # ảnh đó có mang thông tin (không phải ảnh trắng)\n        n_tiles_with_info = (img3.reshape(img3.shape[0],-1).sum(1) < tile_size ** 2 * 3 * 255).sum()\n        # Nếu kích thước ảnh gốc nhỏ kích thước số ảnh con cần chia thì phải pading cho đủ\n        if len(img) < n_tiles:\n            img3 = np.pad(img3,[[0,N-len(img3)],[0,0],[0,0],[0,0]], constant_values=255)\n        idxs = np.argsort(img3.reshape(img3.shape[0],-1).sum(-1))[:n_tiles]\n        img3 = img3[idxs]\n        for i in range(len(img3)):\n            result.append({'img':img3[i], 'idx':i})\n        return result, n_tiles_with_info >= n_tiles\n\nclass PANDADataset(Dataset):\n    def __init__(self,\n                 df,\n                 image_size,\n                 n_tiles=n_tiles,\n                 tile_mode=0,\n                 rand=False,\n                 sub_imgs=False,\n                 transform=None\n                ):\n\n        self.df = df.reset_index(drop=True)\n        self.image_size = image_size\n        self.n_tiles = n_tiles\n        self.tile_mode = tile_mode\n        self.rand = rand\n        self.sub_imgs = sub_imgs\n\n    def __len__(self):\n        return self.df.shape[0]\n\n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        img_id = row.image_id\n        \n        tiff_file = os.path.join(image_folder, f'{img_id}.tiff')\n        #image = skimage.io.MultiImage(tiff_file)[1]\n        image = skimage.io.imread(tiff_file)\n        tiles, OK = get_tiles(image, self.tile_mode)\n\n        if self.rand:\n            idxes = np.random.choice(list(range(self.n_tiles)), self.n_tiles, replace=False)\n        else:\n            idxes = list(range(self.n_tiles))\n        idxes = np.asarray(idxes) + self.n_tiles if self.sub_imgs else idxes\n\n        n_row_tiles = int(np.sqrt(self.n_tiles))\n        images = np.zeros((image_size * n_row_tiles, image_size * n_row_tiles, 3))\n        for h in range(n_row_tiles):\n            for w in range(n_row_tiles):\n                i = h * n_row_tiles + w\n    \n                if len(tiles) > idxes[i]:\n                    this_img = tiles[idxes[i]]['img']\n                else:\n                    this_img = np.ones((self.image_size, self.image_size, 3)).astype(np.uint8) * 255\n                this_img = 255 - this_img\n                h1 = h * image_size\n                w1 = w * image_size\n                images[h1:h1+image_size, w1:w1+image_size] = this_img\n\n#         images = 255 - images\n        images = images.astype(np.float32)\n        images /= 255\n        images = images.transpose(2, 0, 1)\n\n        return torch.tensor(images)","metadata":{"execution":{"iopub.status.busy":"2022-03-02T02:27:44.659203Z","iopub.execute_input":"2022-03-02T02:27:44.659504Z","iopub.status.idle":"2022-03-02T02:27:44.688488Z","shell.execute_reply.started":"2022-03-02T02:27:44.659468Z","shell.execute_reply":"2022-03-02T02:27:44.686731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transforms_train = albumentations.Compose([\n    albumentations.Transpose(p=0.5),\n    albumentations.VerticalFlip(p=0.5),\n    albumentations.HorizontalFlip(p=0.5),\n])\ntransforms_val = albumentations.Compose([])","metadata":{"execution":{"iopub.status.busy":"2022-03-01T13:36:57.936256Z","iopub.execute_input":"2022-03-01T13:36:57.936793Z","iopub.status.idle":"2022-03-01T13:36:57.941119Z","shell.execute_reply.started":"2022-03-01T13:36:57.93676Z","shell.execute_reply":"2022-03-01T13:36:57.940258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_idx = np.where((train['fold'] != fold))[0]\nvalid_idx = np.where((train['fold'] == fold))[0]\n\ndf_this  = train.loc[train_idx]\ndf_valid = train.loc[valid_idx]\n\ndataset_train = PANDADataset(df_this , image_size, n_tiles, transform=transforms_train)\ndataset_valid = PANDADataset(df_valid, image_size, n_tiles, transform=transforms_val)\nprint(len(dataset_train))","metadata":{"execution":{"iopub.status.busy":"2022-03-01T13:54:52.107584Z","iopub.execute_input":"2022-03-01T13:54:52.108342Z","iopub.status.idle":"2022-03-01T13:54:52.119121Z","shell.execute_reply.started":"2022-03-01T13:54:52.108298Z","shell.execute_reply":"2022-03-01T13:54:52.118184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not is_test:\n    dataset_show = PANDADataset(df, image_size, n_tiles, 0)\n    from pylab import rcParams\n    rcParams['figure.figsize'] = 20,10\n    for i in range(2):\n        f, axarr = plt.subplots(1,5)\n        for p in range(5):\n            idx = np.random.randint(0, len(dataset_show))\n            img = dataset_show[idx]\n            axarr[p].imshow(1. - img.transpose(0, 1).transpose(1,2).squeeze())\n            axarr[p].set_title(str(idx))","metadata":{"execution":{"iopub.status.busy":"2022-03-02T02:45:28.74941Z","iopub.execute_input":"2022-03-02T02:45:28.749696Z","iopub.status.idle":"2022-03-02T02:46:33.098812Z","shell.execute_reply.started":"2022-03-02T02:45:28.74967Z","shell.execute_reply":"2022-03-02T02:46:33.097886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_path = os.path.join(BASE_FOLDER+\"train_images\", '005e66f06bce9c2e49142536caf2f6ee.tiff')\nimg = skimage.io.imread(img_path)\n#print(len(img))\nh, w, c = img.shape\npad_h = (tile_size - h % tile_size) % tile_size + ((tile_size * 0) // 2)\npad_w = (tile_size - w % tile_size) % tile_size + ((tile_size * 0) // 2)\n\nimg2 = np.pad(img,[[pad_h // 2, pad_h - pad_h // 2], [pad_w // 2,pad_w - pad_w//2], [0,0]], constant_values=255)\n#print(img2.shape)\n#print(get_tiles(img,0))\nimg3 = img2.reshape(\n    img2.shape[0] // tile_size,\n    tile_size,\n    img2.shape[1] // tile_size,\n   tile_size,\n    3)\n#print(img3.shape)\nimg3 = img3.transpose(0,2,1,3,4).reshape(-1,tile_size,tile_size,3)\nidxs = np.argsort(img3.reshape(img3.shape[0],-1).sum(-1))[:n_tiles]\nimg3 = img3[idxs]\nprint(img3.shape)\nprint(len(img3))\n#print(idxs)\n#print(img3.shape)\n#print(len(img))\n#what = img3.reshape(img3.shape[0],-1).sum(-1)\n#print(what.shape)\n#n_tiles_with_info = (img3.reshape(img3.shape[0],-1).sum(1) < tile_size ** 2 * 3 * 255).sum()\n#print(n_tiles_with_info)","metadata":{"execution":{"iopub.status.busy":"2022-02-23T12:45:01.067523Z","iopub.execute_input":"2022-02-23T12:45:01.067782Z","iopub.status.idle":"2022-02-23T12:45:17.436292Z","shell.execute_reply.started":"2022-02-23T12:45:01.067755Z","shell.execute_reply":"2022-02-23T12:45:17.435539Z"},"trusted":true},"execution_count":null,"outputs":[]}]}