{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install /kaggle/input/staintools-offline/spams-2.6.5.4-cp37-cp37m-linux_x86_64.whl\n!pip install /kaggle/input/staintools-offline/staintools-2.1.2-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:22:15.633663Z","iopub.execute_input":"2022-09-09T18:22:15.634081Z","iopub.status.idle":"2022-09-09T18:23:19.019947Z","shell.execute_reply.started":"2022-09-09T18:22:15.634001Z","shell.execute_reply":"2022-09-09T18:23:19.018845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append('../input/strip-ai-seam-carving/seam-carving')","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:23:21.993718Z","iopub.execute_input":"2022-09-09T18:23:21.994471Z","iopub.status.idle":"2022-09-09T18:23:22.000033Z","shell.execute_reply.started":"2022-09-09T18:23:21.994429Z","shell.execute_reply":"2022-09-09T18:23:21.998717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, random, gc\nimport seaborn as sns\nimport numpy as np \nimport pandas as pd \nimport torch\nimport torch.nn as nn # neural network module\nimport torch.nn.functional as F # neural network module에서 자주 사용되는 함수\nimport torchvision\nimport cv2, math, shutil # OpenCV => cv2\nimport albumentations as Albu\nimport rasterio\nimport staintools\nimport seam_carving\nfrom rasterio.enums import Resampling\nfrom rasterio.transform import Affine\nfrom torchvision import models\nfrom torchvision import transforms\nfrom albumentations.pytorch import ToTensorV2  \nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.metrics import roc_auc_score\nfrom tqdm.notebook import tqdm\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-09T18:23:22.165774Z","iopub.execute_input":"2022-09-09T18:23:22.166417Z","iopub.status.idle":"2022-09-09T18:23:25.008631Z","shell.execute_reply.started":"2022-09-09T18:23:22.166381Z","shell.execute_reply":"2022-09-09T18:23:25.00768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1.1 Test, Submission Path Setting\ndata_path = '../input/mayo-clinic-strip-ai/'\ntest = pd.read_csv(data_path + 'test.csv')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv')\nsubmission_efficient = pd.read_csv(data_path + 'sample_submission.csv')\nsubmission_vit = pd.read_csv(data_path + 'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:29:30.579907Z","iopub.execute_input":"2022-09-09T18:29:30.580886Z","iopub.status.idle":"2022-09-09T18:29:30.597292Z","shell.execute_reply.started":"2022-09-09T18:29:30.580848Z","shell.execute_reply":"2022-09-09T18:29:30.596406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1.1.1 Seed Fasten\nseed = 50\nos.environ['PYTHONHASHSEED'] = str(seed) # python 난수 고정 => 환경변수 접근, 고정 시드 넘버 할당\nrandom.seed(seed) # random module 난수 고정\nnp.random.seed(seed) # numpy module 난수 고정\ntorch.manual_seed(seed)\n\ntorch.manual_seed(seed) # 파이토치 CPU 난수 생성기 => 시드 고정 \ntorch.cuda.manual_seed(seed) # 파이토치 GPU 난수 생성기 => 시드 고정\ntorch.cuda.manual_seed_all(seed) # 파이토치 멀티 코어_GPU 난수 생성기 => 시드 고정\n\ntorch.backends.cudnn.deterministic = True # 확정적 연산 사용\ntorch.backends.cudnn.benchmark = False # benchmark function 해제\ntorch.backends.cudnn.enabled = False # cudnn 사용 해제","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:23:25.038242Z","iopub.execute_input":"2022-09-09T18:23:25.038564Z","iopub.status.idle":"2022-09-09T18:23:25.048458Z","shell.execute_reply.started":"2022-09-09T18:23:25.038525Z","shell.execute_reply":"2022-09-09T18:23:25.047591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1.2.1 Device setting => GPU\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ndevice","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:23:25.053091Z","iopub.execute_input":"2022-09-09T18:23:25.053422Z","iopub.status.idle":"2022-09-09T18:23:25.130919Z","shell.execute_reply.started":"2022-09-09T18:23:25.053394Z","shell.execute_reply":"2022-09-09T18:23:25.129824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dir_path = './test'\nos.mkdir(dir_path)","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:23:25.132896Z","iopub.execute_input":"2022-09-09T18:23:25.133638Z","iopub.status.idle":"2022-09-09T18:23:25.140018Z","shell.execute_reply.started":"2022-09-09T18:23:25.133601Z","shell.execute_reply":"2022-09-09T18:23:25.139096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1.2.2 Test Data Set Convert (Tiff to Png)\nimage_scalar = 1024\nseam_carving.carve.MAX_MEAN_ENERGY = 10.0\nfor i in tqdm(range(test.shape[0])):\n    # Remove Background White Space\n        img_id = test.iloc[i].image_id\n        img_path = f'../input/mayo-clinic-strip-ai/test/{img_id}.tif'\n\n        image = rasterio.open(img_path)\n        image = image.read(out_shape=(image.count, int(image_scalar), int(image_scalar)),\n                                 resampling=Resampling.bilinear).transpose(1,2,0)\n        image_h, image_w, _ = image.shape\n        image = seam_carving.resize(image, (image_w-512, image_h-512),\n                                        energy_mode='backward',\n                                        order=('width-first'),\n                                        keep_mask=None)\n    # Apply Stain Color Normalization\n        if img_id == '006388_0':\n            target_image = image\n            image = image.transpose(2,0,1)\n            with rasterio.open(f'./test/{img_id}.png', 'w', driver='png', height = image.shape[1], # driver => jpeg, png, gtiff\n                           width = image.shape[2], dtype = image.dtype, count=3) as images: # count => Image Channel 개수 (RGB의 경우 3개)\n                images.write(image)\n\n            continue\n        \n        else:\n            target_img = staintools.LuminosityStandardizer.standardize(target_image)\n            transform_img = staintools.LuminosityStandardizer.standardize(image)\n\n            normalizer = staintools.StainNormalizer(method='vahadane')\n            target_img_n = normalizer.fit(target_img)\n            transform_img_n = normalizer.transform(transform_img)\n            image = transform_img_n.transpose(2,0,1)\n            with rasterio.open(f'./test/{img_id}.png', 'w', driver='png', height = image.shape[1], # driver => jpeg, png, gtiff\n                       width = image.shape[2], dtype = image.dtype, count=3) as images: # count => Image Channel 개수 (RGB의 경우 3개)\n                images.write(image)\n\n            del image\n            gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:23:25.7256Z","iopub.execute_input":"2022-09-09T18:23:25.72595Z","iopub.status.idle":"2022-09-09T18:27:46.009197Z","shell.execute_reply.started":"2022-09-09T18:23:25.72592Z","shell.execute_reply":"2022-09-09T18:27:46.008182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1.2.2 Definition Data Set Class\nclass ImageDataset(Dataset):\n    def __init__(self, df, img_dir='./', transform=None, is_test=True):\n        super().__init__()\n        self.df = df\n        self.img_dir = img_dir\n        self.transform = transform\n        self.is_test = is_test\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n    \n        if self.is_test:\n            img_id = self.df.iloc[idx, 0] \n            img_path = f'{self.img_dir}/{img_id}.png'\n            \n            image = rasterio.open(img_path)\n            image = image.read(resampling=Resampling.bilinear).transpose(1,2,0)\n            patient_ids = self.df.iloc[idx, 2]\n            \n            if self.transform is not None:\n              image = self.transform(image=image)['image']\n            \n            return image, patient_ids ","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:27:55.014395Z","iopub.execute_input":"2022-09-09T18:27:55.014842Z","iopub.status.idle":"2022-09-09T18:27:55.025694Z","shell.execute_reply.started":"2022-09-09T18:27:55.014767Z","shell.execute_reply":"2022-09-09T18:27:55.024672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1.3 Test Transform\ntransform_test = Albu.Compose([Albu.Normalize(mean=[0.485, 0.456, 0.406], \n                                              std=[0.229, 0.224, 0.225]),\n                               ToTensorV2()])","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:27:59.418569Z","iopub.execute_input":"2022-09-09T18:27:59.419014Z","iopub.status.idle":"2022-09-09T18:27:59.426179Z","shell.execute_reply.started":"2022-09-09T18:27:59.418974Z","shell.execute_reply":"2022-09-09T18:27:59.425313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1.4 Test Data Set\nimg_dir = './test'\ndataset_test = ImageDataset(test, img_dir=img_dir, transform=transform_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:27:59.753051Z","iopub.execute_input":"2022-09-09T18:27:59.753488Z","iopub.status.idle":"2022-09-09T18:27:59.761234Z","shell.execute_reply.started":"2022-09-09T18:27:59.753447Z","shell.execute_reply":"2022-09-09T18:27:59.759849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1.5 Data Loader Definition \ndef seed_worker(worker_id):\n    worker_seed = torch.initial_seed() %2**32\n    np.random.seed(worker_seed)\n    random.seed(worker_seed)\n\ng = torch.Generator()\ng.manual_seed(0)\n\nbatch_size = 1\nloader_test = DataLoader(dataset_test, batch_size=batch_size, shuffle=False, \n                         worker_init_fn=seed_worker, generator=g, num_workers=1)","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:28:00.91792Z","iopub.execute_input":"2022-09-09T18:28:00.918374Z","iopub.status.idle":"2022-09-09T18:28:00.926116Z","shell.execute_reply.started":"2022-09-09T18:28:00.918335Z","shell.execute_reply":"2022-09-09T18:28:00.925157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1.6 Pretrained Model => Efficient Net-B5\n# 앙상블 추후 적용 => 모델 리스트화\nmodel = []\n\n# Efficient-Net\nefficient_net = models.efficientnet_b0()\nefficient_net.classifier = nn.Linear(1280,2)\nefficient_net.load_state_dict(torch.load('../input/strip-ai-train-parameters/new_efficient_model_state_dict.pth'))\nprint(efficient_net.classifier)\n\nefficient_net = efficient_net.to(device)\nmodel.append(efficient_net)","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:28:06.690149Z","iopub.execute_input":"2022-09-09T18:28:06.69052Z","iopub.status.idle":"2022-09-09T18:28:10.013985Z","shell.execute_reply.started":"2022-09-09T18:28:06.690487Z","shell.execute_reply":"2022-09-09T18:28:10.012724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# regnet_x = models.regnet_x_8gf()\n# regnet_x.fc = nn.Linear(1920,2)\n# regnet_x.load_state_dict(torch.load('../input/strip-ai-regnet-train-1k-png/model_state_dict.pth'))\n# print(regnet_x.fc)\n\n# regnet_x = regnet_x.to(device)\n# model.append(regnet_x)","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:28:10.015844Z","iopub.execute_input":"2022-09-09T18:28:10.016292Z","iopub.status.idle":"2022-09-09T18:28:10.021564Z","shell.execute_reply.started":"2022-09-09T18:28:10.016256Z","shell.execute_reply":"2022-09-09T18:28:10.020514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vit = models.vit_l_16(image_size=512)\nvit.heads = nn.Linear(1024,2)\nvit.load_state_dict(torch.load('../input/strip-ai-vit-train-parameter/vit_model_state_dict.pth'))\n\nvit = vit.to(device)\nmodel.append(vit)\n# i want to test Vision Transformer model but, computing powers that i have are limited.... ","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:28:10.023105Z","iopub.execute_input":"2022-09-09T18:28:10.024172Z","iopub.status.idle":"2022-09-09T18:28:24.839105Z","shell.execute_reply.started":"2022-09-09T18:28:10.024135Z","shell.execute_reply":"2022-09-09T18:28:24.838084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1.7 Predict Function Definition\ndef predict(model, loader_test): \n    ids = []\n    model.eval()\n    preds_part = []\n    \n    with torch.no_grad():\n        for images, patient_ids in loader_test:\n            images = images.to(device) #, dtype=torch.float32)\n            outputs = model(images)\n            \n            ids.append(patient_ids[0]) # Pytorch Broadcasting => patient_ids 차원 추가, 그래서 ('patient_id',) <= 출력 형태가 저런 모습이었다. \n            preds_part.append(torch.softmax(outputs.cpu(), dim=1).squeeze().numpy())\n            \n    return np.array(preds_part), ids","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:28:28.113522Z","iopub.execute_input":"2022-09-09T18:28:28.114339Z","iopub.status.idle":"2022-09-09T18:28:28.125062Z","shell.execute_reply.started":"2022-09-09T18:28:28.114289Z","shell.execute_reply":"2022-09-09T18:28:28.123752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.8.6 Submission\npreds, ids = predict(model[0], loader_test)\nresult_table_efficient = pd.DataFrame({'patient_id' : ids, \"CE\" : preds[:,0], \"LAA\" : preds[:,1]})\nresult_table_efficient = result_table_efficient.groupby(\"patient_id\").mean()","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:28:30.626517Z","iopub.execute_input":"2022-09-09T18:28:30.6269Z","iopub.status.idle":"2022-09-09T18:28:31.883994Z","shell.execute_reply.started":"2022-09-09T18:28:30.626866Z","shell.execute_reply":"2022-09-09T18:28:31.882358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_table_efficient","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:28:36.408225Z","iopub.execute_input":"2022-09-09T18:28:36.408872Z","iopub.status.idle":"2022-09-09T18:28:36.432397Z","shell.execute_reply.started":"2022-09-09T18:28:36.408803Z","shell.execute_reply":"2022-09-09T18:28:36.431486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_efficient.CE = result_table_efficient.CE.to_list()\nsubmission_efficient.LAA = result_table_efficient.LAA.to_list()\n\n#submission_efficient[['patient_id', 'CE', 'LAA']].round(6).to_csv('submission.csv', index=False)\nsubmission_efficient","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:28:39.028293Z","iopub.execute_input":"2022-09-09T18:28:39.029465Z","iopub.status.idle":"2022-09-09T18:28:39.04585Z","shell.execute_reply.started":"2022-09-09T18:28:39.029421Z","shell.execute_reply":"2022-09-09T18:28:39.044021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds, ids = predict(model[1], loader_test)\nresult_table_vit = pd.DataFrame({'patient_id' : ids, 'CE' : preds[:,0], 'LAA' : preds[:,1]})\nresult_table_vit = result_table_vit.groupby(\"patient_id\").mean()","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:28:42.239411Z","iopub.execute_input":"2022-09-09T18:28:42.239762Z","iopub.status.idle":"2022-09-09T18:28:42.991267Z","shell.execute_reply.started":"2022-09-09T18:28:42.239731Z","shell.execute_reply":"2022-09-09T18:28:42.990101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_table_vit","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:28:45.157793Z","iopub.execute_input":"2022-09-09T18:28:45.158437Z","iopub.status.idle":"2022-09-09T18:28:45.169582Z","shell.execute_reply.started":"2022-09-09T18:28:45.158396Z","shell.execute_reply":"2022-09-09T18:28:45.16846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_vit.CE = result_table_vit.CE.to_list()\nsubmission_vit.LAA = result_table_vit.LAA.to_list()\nsubmission_vit","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:29:36.016352Z","iopub.execute_input":"2022-09-09T18:29:36.017203Z","iopub.status.idle":"2022-09-09T18:29:36.034286Z","shell.execute_reply.started":"2022-09-09T18:29:36.017159Z","shell.execute_reply":"2022-09-09T18:29:36.033034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.CE = (submission_efficient.CE + submission_vit.CE) / 2\nsubmission.LAA = (submission_efficient.LAA + submission_vit.LAA) / 2\n\nsubmission[['patient_id', 'CE', 'LAA']].round(6).to_csv('submission.csv', index=False) # CE, LAA만 괄호에 넣었더니 patient_id이 사라진 상태로 CSV 파일이 저장된 경우....\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-09-09T18:30:03.535555Z","iopub.execute_input":"2022-09-09T18:30:03.535941Z","iopub.status.idle":"2022-09-09T18:30:03.556177Z","shell.execute_reply.started":"2022-09-09T18:30:03.535908Z","shell.execute_reply":"2022-09-09T18:30:03.555281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-09-09T18:30:47.661279Z","iopub.execute_input":"2022-09-09T18:30:47.66164Z","iopub.status.idle":"2022-09-09T18:30:48.722564Z","shell.execute_reply.started":"2022-09-09T18:30:47.661609Z","shell.execute_reply":"2022-09-09T18:30:48.721433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}