{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**If you want to see EDA & Data Preprocessing, Click the below links!!**\n\n**1) EDA & Convert Original Train Data to \"1K PNG Image\" =>** https://www.kaggle.com/code/qcqced/strip-ai-eda-data-preprocessing-1k-png\n\n**2) Convert \"1K PNG Image\" to \"Remove Unnecessary Background Image\" =>** https://www.kaggle.com/code/qcqced/strip-ai-remove-background-1k-png-image","metadata":{}},{"cell_type":"markdown","source":"**If you want to see this code's scoring result, click this link (Efficient-Net & RegNet Ensemble)**\n\nhttps://www.kaggle.com/code/qcqced/strip-ai-submission-pipeline/notebook?scriptVersionId=103752622","metadata":{}},{"cell_type":"markdown","source":"**I Found very very Low quality of Train Data, so except them in Training session**\n\n**(23b492_0, 280c26_0, e26a04_0, e72352_0)**","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport numpy as np \nimport pandas as pd \nimport os, sys, random, gc\nimport torch\nimport torch.nn as nn # neural network module\nimport torch.nn.functional as F # neural network module에서 자주 사용되는 함수\nimport torchvision\nimport matplotlib.pyplot as plt\nimport matplotlib as matp\nimport matplotlib.gridspec as gridspec \nimport cv2, math, shutil # OpenCV => cv2\nimport albumentations as Albu\nfrom torchvision import models\nfrom torchvision import transforms\nfrom albumentations.pytorch import ToTensorV2  \nfrom sklearn.model_selection import train_test_split\nfrom zipfile import ZipFile\nfrom torch.utils.data import Dataset, DataLoader, WeightedRandomSampler\nfrom sklearn.metrics import roc_auc_score\nfrom tqdm.notebook import tqdm \nfrom transformers import get_cosine_schedule_with_warmup # 스케줄러\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-28T02:47:38.864701Z","iopub.execute_input":"2022-09-28T02:47:38.865163Z","iopub.status.idle":"2022-09-28T02:47:46.645868Z","shell.execute_reply.started":"2022-09-28T02:47:38.865072Z","shell.execute_reply":"2022-09-28T02:47:46.644929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1. Data Upload & Check\n# Goal of Competition => 허혈성 뇌졸증의 원인이 되는 두 가지 혈전증을 병리적 이미지를 통해 분류\n# Evaluation of Competition => Binary Classification\n# CE => cardioembolic, 심인성 색전증\n# LAA => Large artery atherosclerosis, 큰동맥죽상경화증\ndata_path = '../input/mayo-clinic-strip-ai/'\n\nlabels = pd.read_csv(data_path + 'train.csv') # Train Data Set\ntest = pd.read_csv(data_path + 'test.csv')\nsubmission = pd.read_csv(data_path + 'sample_submission.csv')\n\nlabels, test, submission ","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:50.152623Z","iopub.execute_input":"2022-09-28T02:47:50.152995Z","iopub.status.idle":"2022-09-28T02:47:50.177028Z","shell.execute_reply.started":"2022-09-28T02:47:50.152967Z","shell.execute_reply":"2022-09-28T02:47:50.175912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1.2 Remove Bad Quality Train data\n# carving_remove_list = ['23b492_0', '280c26_0', 'e26a04_0', 'e72352_0', '0ff890_0',\n#                'c31442_0', 'fd3079_0']\n# remove_list = [‘4094c4_0’, ‘f58fcb_0’, ‘5ae30a_0’, ‘41db5b_0’, ‘a59c0d_0’, ‘e8dceb_0’, ‘437435_1’, ‘062387_0’,\n# ‘70c523_0’, ‘82653f_0’, ‘a59c0d_1’, ‘c12836_1’, ‘36a149_0’, ‘48af1a_0’, ‘055f6a_0’, ‘33da32_0’,\n#  ‘140d1a_0’, ‘c12836_0’, ‘b43ebe_1’, ‘46def4_0’, ‘dd05a2_0’, ‘8b3323_0’, ‘652471_0’, ‘4dfe1e_0’,\n#  ‘029c68_0’, ‘112b6e_0’, ‘b4a34d_0’, ‘4919a7_0’, ‘70c523_2’, ‘0fcfe9_0’, ‘ff14e0_0’, ‘41b4ea_0’,\n#  ‘50ba15_0’, ‘4f9ac6_0’, ‘dd6316_0’, ‘41b4ea_1’, ‘579988_0’, ‘ec3098_0’, ‘41b23c_0’, ‘da732b_0’,\n#  ‘72df9e_1’, ‘437435_0’, ‘0aaeb3_0’, ‘70c523_1’, ‘4667f6_0’, ‘9fd56b_0’, ‘c78d6b_0’, ‘86e319_0’,\n#  ‘caf901_0’, ‘65fe16_0’, ‘56d177_4’, ‘56d177_3’, ‘0ff890_0’, ‘82399d_0’, ‘56d177_0’]\n# for i in range(len(remove_list)):\n#     remove_idx = labels[labels['image_id'] == remove_list[i]].index\n#     labels = labels.drop(remove_idx)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:50.327495Z","iopub.execute_input":"2022-09-28T02:47:50.327835Z","iopub.status.idle":"2022-09-28T02:47:50.33314Z","shell.execute_reply.started":"2022-09-28T02:47:50.327807Z","shell.execute_reply":"2022-09-28T02:47:50.332047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:50.475346Z","iopub.execute_input":"2022-09-28T02:47:50.4757Z","iopub.status.idle":"2022-09-28T02:47:50.494613Z","shell.execute_reply.started":"2022-09-28T02:47:50.475672Z","shell.execute_reply":"2022-09-28T02:47:50.493326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Center_id \nlabels['center_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:50.618158Z","iopub.execute_input":"2022-09-28T02:47:50.618526Z","iopub.status.idle":"2022-09-28T02:47:50.632152Z","shell.execute_reply.started":"2022-09-28T02:47:50.618497Z","shell.execute_reply":"2022-09-28T02:47:50.630898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Center_id \nlabels['center_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:50.801465Z","iopub.execute_input":"2022-09-28T02:47:50.802389Z","iopub.status.idle":"2022-09-28T02:47:50.809993Z","shell.execute_reply.started":"2022-09-28T02:47:50.802358Z","shell.execute_reply":"2022-09-28T02:47:50.809075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label => (CE, LAA) == (547, 207)\nlabels.loc[labels['label'] == \"CE\"], len(labels.loc[labels['label'] == \"CE\"]) # CE => 547개\nlabels.loc[labels['label'] == \"LAA\"], len(labels.loc[labels['label'] == \"LAA\"]) # LAA => 207개","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:51.510372Z","iopub.execute_input":"2022-09-28T02:47:51.511246Z","iopub.status.idle":"2022-09-28T02:47:51.527127Z","shell.execute_reply.started":"2022-09-28T02:47:51.511187Z","shell.execute_reply":"2022-09-28T02:47:51.526156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=labels, x='label')","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:51.636107Z","iopub.execute_input":"2022-09-28T02:47:51.636838Z","iopub.status.idle":"2022-09-28T02:47:51.887715Z","shell.execute_reply.started":"2022-09-28T02:47:51.636806Z","shell.execute_reply":"2022-09-28T02:47:51.886717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Label.value / 전체 모수 => (CE, LAA) == (72.5%, 27.5%)\n# labels.label.value_counts() / labels.shape[0] => 이 코드로 대체 가능.\nlen(labels.loc[labels['label'] == \"CE\"]) / len(labels['label']), len(labels.loc[labels['label'] == \"LAA\"]) / len(labels['label']) ","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:51.892305Z","iopub.execute_input":"2022-09-28T02:47:51.894572Z","iopub.status.idle":"2022-09-28T02:47:51.907771Z","shell.execute_reply.started":"2022-09-28T02:47:51.894534Z","shell.execute_reply":"2022-09-28T02:47:51.906721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.patient_id.nunique() # 627명의 환자가 1개의 병리적 이미지를 가지고 있음 => 한 명의 환자가 CE, LAA에 모두 해당되는 Case가 존재할까??\nplt.figure(figsize=(10,5))\nsns.countplot(labels.groupby(\"patient_id\").image_num.size(), palette=\"Greens_r\")\nplt.xlabel(\"Number of images per patient\")\nplt.title(\"Max image number per patient in train\")","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:51.938221Z","iopub.execute_input":"2022-09-28T02:47:51.940538Z","iopub.status.idle":"2022-09-28T02:47:52.140143Z","shell.execute_reply.started":"2022-09-28T02:47:51.940503Z","shell.execute_reply":"2022-09-28T02:47:52.139261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.groupby(\"patient_id\").label.nunique().max() # 그런 환자는 없는 것으로 확인, 단순히 한 명의 환자가 가진 이미지가 여러 장일 수 있는 것.","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:54.8359Z","iopub.execute_input":"2022-09-28T02:47:54.836297Z","iopub.status.idle":"2022-09-28T02:47:54.847908Z","shell.execute_reply.started":"2022-09-28T02:47:54.836266Z","shell.execute_reply":"2022-09-28T02:47:54.846866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Center와 Target의 관계성 => 특별한 관계는 없는 듯...?\nsns.set_style(style=\"darkgrid\")\nplt.figure(figsize=(15,8))\n\ngraph = sns.histplot(data=labels, x='center_id', hue='label', multiple='dodge', \n                     discrete=True, kde = True)\n\ngraph.set(xlim=(0,12), xticks=np.arange(0,12,1)) # x축 간격 설정\ngraph.set(ylim=(0,195), yticks=np.arange(0,195,15))\ngraph.set(xlabel=\"Center ID\", ylabel=\"Label Count\")\ngraph.set_title('Distribution of Labels by Center', fontsize=16)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:54.962714Z","iopub.execute_input":"2022-09-28T02:47:54.963072Z","iopub.status.idle":"2022-09-28T02:47:55.342242Z","shell.execute_reply.started":"2022-09-28T02:47:54.963043Z","shell.execute_reply":"2022-09-28T02:47:55.341255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 2. Target Data Check \nnum_img = 4\n\nCE = labels.loc[labels['label'] == \"CE\"]\nLAA = labels.loc[labels['label'] == \"LAA\"]\n\nlast_CE_img_id = CE['image_id'][-num_img:]\nlast_LAA_img_id = LAA['image_id'][-num_img:]\n\nCE_img_id = CE['image_id']\nLAA_img_id = LAA['image_id']\nlast_CE_img_id","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:55.858453Z","iopub.execute_input":"2022-09-28T02:47:55.858848Z","iopub.status.idle":"2022-09-28T02:47:55.871295Z","shell.execute_reply.started":"2022-09-28T02:47:55.858816Z","shell.execute_reply":"2022-09-28T02:47:55.870263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image Visualization Function => CE와 LAA는 어떤 차이가 있을까??\nimport rasterio\nfrom rasterio.enums import Resampling\nfrom rasterio.transform import Affine\n\nimage_scalar = 0.1\n\ndef img_show(img_ids, rows=2, cols=2): # diseases_name => string type\n    matp.rc('font', size=10)\n    plt.figure(figsize=(40,40))\n    grid = gridspec.GridSpec(rows, cols)\n    \n    for idx, img_id in enumerate(img_ids):\n        img_path = f'{data_path}/train/{img_id}.tif'\n        img = rasterio.open(img_path)\n        image = img.read(out_shape=(img.count, int(img.height * image_scalar), int(img.width * image_scalar)),\n                         resampling=Resampling.bilinear).transpose(1,2,0) # imshow() => (너비, 높이, 채널) 순으로 매개변수를 요구하기 때문에 Transpose 필요함\n        print(image.shape)\n        print(type(image))\n        ax = plt.subplot(grid[idx])\n        ax.imshow(image)\n        \n        gc.collect()\n        del img, image","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:56.067871Z","iopub.execute_input":"2022-09-28T02:47:56.068472Z","iopub.status.idle":"2022-09-28T02:47:56.28995Z","shell.execute_reply.started":"2022-09-28T02:47:56.068435Z","shell.execute_reply":"2022-09-28T02:47:56.28905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_show(last_CE_img_id)","metadata":{"execution":{"iopub.status.busy":"2022-09-09T17:10:23.532441Z","iopub.execute_input":"2022-09-09T17:10:23.533277Z","iopub.status.idle":"2022-09-09T17:10:38.228409Z","shell.execute_reply.started":"2022-09-09T17:10:23.533241Z","shell.execute_reply":"2022-09-09T17:10:38.227052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_show(last_LAA_img_id)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('../input/strip-ai-remove-bad-image-2k-skimage-stain/mayo_train.csv')\nlabels","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:57.825007Z","iopub.execute_input":"2022-09-28T02:47:57.825495Z","iopub.status.idle":"2022-09-28T02:47:57.858152Z","shell.execute_reply.started":"2022-09-28T02:47:57.825456Z","shell.execute_reply":"2022-09-28T02:47:57.8571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.1 Seed Fasten\nseed = 50\nos.environ['PYTHONHASHSEED'] = str(seed) # python 난수 고정 => 환경변수 접근, 고정 시드 넘버 할당\nrandom.seed(seed) # random module 난수 고정\nnp.random.seed(seed) # numpy module 난수 고정\ntorch.manual_seed(seed)\n\ntorch.manual_seed(seed) # 파이토치 CPU 난수 생성기 => 시드 고정 \ntorch.cuda.manual_seed(seed) # 파이토치 GPU 난수 생성기 => 시드 고정\ntorch.cuda.manual_seed_all(seed) # 파이토치 멀티 코어_GPU 난수 생성기 => 시드 고정\n\ntorch.backends.cudnn.deterministic = True # 확정적 연산 사용\ntorch.backends.cudnn.benchmark = False # benchmark function 해제\ntorch.backends.cudnn.enabled = False # cudnn 사용 해제","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:57.949299Z","iopub.execute_input":"2022-09-28T02:47:57.949828Z","iopub.status.idle":"2022-09-28T02:47:57.961831Z","shell.execute_reply.started":"2022-09-28T02:47:57.949782Z","shell.execute_reply":"2022-09-28T02:47:57.960748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.2 Device setting => GPU\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ndevice","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:47:58.104855Z","iopub.execute_input":"2022-09-28T02:47:58.105262Z","iopub.status.idle":"2022-09-28T02:47:58.194231Z","shell.execute_reply.started":"2022-09-28T02:47:58.105188Z","shell.execute_reply":"2022-09-28T02:47:58.193107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.3 train/valid split\ntrain, valid = train_test_split(labels, test_size=0.1, stratify=labels['label'], random_state=50)\ntrain['label'].value_counts() / len(train.label), labels['label'].value_counts() / len(labels.label) # 모수와 동일한 비율로 데이터 세트 분할된지 확인","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:48:15.575555Z","iopub.execute_input":"2022-09-28T02:48:15.575916Z","iopub.status.idle":"2022-09-28T02:48:15.591794Z","shell.execute_reply.started":"2022-09-28T02:48:15.575886Z","shell.execute_reply":"2022-09-28T02:48:15.590547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train/valid 데이터 세트 총합 == 모수 확인\ntrain, valid","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:48:15.74706Z","iopub.execute_input":"2022-09-28T02:48:15.74776Z","iopub.status.idle":"2022-09-28T02:48:15.764855Z","shell.execute_reply.started":"2022-09-28T02:48:15.747724Z","shell.execute_reply":"2022-09-28T02:48:15.763772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.4 Data Class Definition\nclass ImageDataset(Dataset):\n    def __init__(self, df, img_dir='./', transform=None, is_test=False):\n        super().__init__()\n        self.df = df\n        self.img_dir = img_dir\n        self.transform = transform\n        self.is_test = is_test\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        if self.is_test:\n            img_id = self.df.iloc[idx, 0] \n            img_path = f'{self.img_dir}/{img_id}.tif'\n            \n            image = rasterio.open(img_path)\n            image = image.read(out_shape=(3, int(512), int(512)),\n                              resampling=Resampling.bilinear).transpose(1,2,0)\n            patient_ids = self.df.iloc[idx, 2]\n            \n            if self.transform is not None:\n                image = self.transform(image=image)['image']                \n            return image, patient_ids # image만 반환 => test의 경우 label X\n        \n        else:\n            img_id = self.df.iloc[idx, 0]\n            label = self.df.iloc[idx, 4] \n            \n            img_path = f'{self.img_dir}/{img_id}.png'\n#             img_path = f'{self.img_dir}/{img_id}.jpg'\n\n            # Apply Skimage Stain Normailize => Not Using CV2 BRG2RGB\n            image = cv2.imread(img_path)\n#             image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n#             image = rasterio.open(img_path)\n#             image = image.read(resampling=Resampling.bilinear).transpose(1,2,0)\n#             image = image.astype(np.float32) \n            \n            # 이미지 변환기 사용 여부 세팅\n            if self.transform is not None:\n                image = self.transform(image=image)['image']\n            \n            \n            label_dict = {\"CE\" : 0, \"LAA\" : 1}\n            label = label_dict[label]\n            return image, label # 둘 다 반환","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:48:15.911702Z","iopub.execute_input":"2022-09-28T02:48:15.912381Z","iopub.status.idle":"2022-09-28T02:48:15.923534Z","shell.execute_reply.started":"2022-09-28T02:48:15.912341Z","shell.execute_reply":"2022-09-28T02:48:15.922459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.5 Data Augmentation\n# train transform\ntransform_train = Albu.Compose([\n                                Albu.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n#                                 Albu.VerticalFlip(p=0.3), # p => 확률을 의미. 0.2는 20%로 적용한다는 뜻\n#                                 Albu.HorizontalFlip(p=0.3),\n                                Albu.ShiftScaleRotate(shift_limit=0.2,\n                                                      scale_limit=0.2,\n                                                      rotate_limit=30, p=0.2),\n                                #Albu.HueSaturationValue(hue_shift_limit=0.2, sat_shift_limit=0.2, val_shift_limit=0.2, p=0.5), # HueSaturationvalue => HSV (색상표현 방식), RGB 보다 사실적인 이미지를 표현하는데 적합\n#                                 Albu.PiecewiseAffine(p=0.2),\n                                ToTensorV2()])\n# valid, test transform\ntransform_test = Albu.Compose([\n                               Albu.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n                               ToTensorV2()])","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:48:37.583069Z","iopub.execute_input":"2022-09-28T02:48:37.584047Z","iopub.status.idle":"2022-09-28T02:48:37.591786Z","shell.execute_reply.started":"2022-09-28T02:48:37.584011Z","shell.execute_reply":"2022-09-28T02:48:37.590573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From last 2 times train test, model (Efficient-Net) has trend that easily overfitting to digtal pathology data.\n\nI think using common Data Augment Policy is right to train model to this pathology data.","metadata":{}},{"cell_type":"code","source":"# Step 3.6 Data Set Init\n# img_dir = '../input/strip-ai-skimage-stain-normalizetrain/train'\n# img_dir = '../input/strip-ai-2k-train-data-set/convert_train'\nimg_dir = '../input/strip-ai-remove-bad-image-2k-skimage-stain/skimage_stain_train'\n\ndataset_train = ImageDataset(train, img_dir=img_dir, transform=transform_train)\ndataset_valid = ImageDataset(valid, img_dir=img_dir, transform=transform_test)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:48:39.13301Z","iopub.execute_input":"2022-09-28T02:48:39.133379Z","iopub.status.idle":"2022-09-28T02:48:39.138527Z","shell.execute_reply.started":"2022-09-28T02:48:39.133348Z","shell.execute_reply":"2022-09-28T02:48:39.137576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.6.1 Weighted Random Sample\nlabel_dict = {\"CE\" : 0, \"LAA\" : 1}\nlabel_number = train['label'].value_counts().to_list()\ntotal_number = sum(label_number)\nlabels = train['label'].to_list()\n\nclass_weights = [total_number / label_number[i] for i in range(len(label_number))]\nweights = [class_weights[label_dict[labels[i]]] for i in range(int(total_number))] #해당 레이블마다의 가중치 비율\nsampler = WeightedRandomSampler(torch.DoubleTensor(weights), int(total_number), replacement=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:48:40.513464Z","iopub.execute_input":"2022-09-28T02:48:40.513821Z","iopub.status.idle":"2022-09-28T02:48:40.523519Z","shell.execute_reply.started":"2022-09-28T02:48:40.513792Z","shell.execute_reply":"2022-09-28T02:48:40.521178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.7 Data Loader Definition => Multi-Processing 활용 \ndef seed_worker(worker_id):\n    worker_seed = torch.initial_seed() %2**32\n    np.random.seed(worker_seed)\n    random.seed(worker_seed)\n\n    \ng = torch.Generator()\ng.manual_seed(0)\n\nbatch_size = 2\nloader_train = DataLoader(dataset_train, batch_size=batch_size, shuffle=True, worker_init_fn=seed_worker,\n                          generator=g, num_workers=2)# num_workers => max 4\nloader_valid = DataLoader(dataset_valid, batch_size=batch_size, shuffle=False, worker_init_fn=seed_worker,\n                          generator=g, num_workers=2)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:48:50.453998Z","iopub.execute_input":"2022-09-28T02:48:50.454374Z","iopub.status.idle":"2022-09-28T02:48:50.463566Z","shell.execute_reply.started":"2022-09-28T02:48:50.454343Z","shell.execute_reply":"2022-09-28T02:48:50.462622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.8.1 Pretrained Model => Efficient Net-B5\n# Very Light & Fast\nmodel = []\nefficientnet = models.efficientnet_b0(pretrained=True)\n#efficientnet.load_state_dict(torch.load('../input/efficient-netb05/efficientnet_b5_lukemelas-b6417697.pth')) # Internet Access X => 오프라인으로 모델 불러오기\nefficientnet.classifier = nn.Linear(1280, 2)\n\nprint(efficientnet.classifier)\nefficientnet = efficientnet.to(device)\n\nmodel.append(efficientnet)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:48:53.195759Z","iopub.execute_input":"2022-09-28T02:48:53.196122Z","iopub.status.idle":"2022-09-28T02:49:35.721051Z","shell.execute_reply.started":"2022-09-28T02:48:53.196092Z","shell.execute_reply":"2022-09-28T02:49:35.720033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pretrained Model 2 => RegNet y 400mf\n# Very Light & Fast \nregnet_y = models.regnet_y_400mf(pretrained=\"IMAGENET1K_V2\")\nregnet_y.fc = nn.Linear(440,2)\nprint(regnet_y.fc)\nregnet_y = regnet_y.to(device)\nmodel.append(regnet_y)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:49:35.723073Z","iopub.execute_input":"2022-09-28T02:49:35.723458Z","iopub.status.idle":"2022-09-28T02:49:37.134007Z","shell.execute_reply.started":"2022-09-28T02:49:35.723421Z","shell.execute_reply":"2022-09-28T02:49:37.132977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# vit = models.vit_l_16(image_size=512)\n# vit.heads = nn.Linear(1024,2)\n\n# vit = vit.to(device)\n# model.append(vit)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:49:37.135387Z","iopub.execute_input":"2022-09-28T02:49:37.135847Z","iopub.status.idle":"2022-09-28T02:49:37.140661Z","shell.execute_reply.started":"2022-09-28T02:49:37.135812Z","shell.execute_reply":"2022-09-28T02:49:37.1396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.8.2 Loss_Function & Optimizer Setting\ncriterion = nn.CrossEntropyLoss() \noptimizer1 = torch.optim.AdamW(model[0].parameters(), lr=0.0003, weight_decay=0.0001) # Efficient-Net\noptimizer2 = torch.optim.AdamW(model[1].parameters(), lr=0.00006, weight_decay=0.0001) \n#optimizer2 = torch.optim.AdamW(model[1].parameters(), lr=0.0003, weight_decay=0.0001) ","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:49:37.143476Z","iopub.execute_input":"2022-09-28T02:49:37.144012Z","iopub.status.idle":"2022-09-28T02:49:37.153658Z","shell.execute_reply.started":"2022-09-28T02:49:37.143977Z","shell.execute_reply":"2022-09-28T02:49:37.152748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.8.3 epoch, scheduler setting \nepochs_efficient = 49\nepochs_regnet = 29\nscheduler1 = get_cosine_schedule_with_warmup(optimizer1, num_warmup_steps=len(loader_train)*3, num_training_steps=len(loader_train)*epochs_efficient)  \nscheduler2 = get_cosine_schedule_with_warmup(optimizer2, num_warmup_steps=len(loader_train)*3, num_training_steps=len(loader_train)*epochs_regnet)  ","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:49:37.154983Z","iopub.execute_input":"2022-09-28T02:49:37.155444Z","iopub.status.idle":"2022-09-28T02:49:37.167146Z","shell.execute_reply.started":"2022-09-28T02:49:37.155411Z","shell.execute_reply":"2022-09-28T02:49:37.166154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.8.4 Model Train & Validation Function Definition => Efficient-Net B05\ndef train(criterion, optimizer, model, loader_train, loader_valid, epochs, scheduler, save_parameter):\n    valid_loss_min = np.inf # Validation Loss 초기화\n    roc_auc_min = -np.inf\n    column = ['Train', 'Validation']\n    df = pd.DataFrame(columns=column) # 훈련 vs 검증 데이터 손실값 비교 그래프 용 데이터프레임\n    \n    for epoch in range(epochs):\n        model.train()\n        epoch_loss = 0\n        \n        # Model Train per epoch\n        for images, labels, in tqdm(loader_train):\n            images = images.to(device)\n            labels = labels.to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            \n            epoch_loss += loss.item()\n            loss.backward()\n            optimizer.step()\n            \n            if scheduler is not None:\n                scheduler.step()\n         \n        print(f'\\t에폭[{epoch+1}/{epochs}] - 손실값: {epoch_loss/len(loader_train):.4f}')   \n        \n        # Model Validation per epoch\n        epoch_valid_loss = 0\n        preds_list = []\n        true_list = []\n\n        model.eval()\n        with torch.no_grad():\n            for images, labels in loader_valid:\n                images = images.to(device)\n                labels = labels.to(device)\n\n                outputs = model(images)\n                loss = criterion(outputs, labels)\n                epoch_valid_loss += loss.item()\n\n                preds = torch.softmax(outputs.cpu(), dim=1).numpy()\n                true = torch.eye(2)[labels].cpu().numpy()\n\n                preds_list.extend(preds) \n                true_list.extend(true)\n\n        roc_auc = roc_auc_score(true_list, preds_list)\n        print(f'\\t검증 데이터 손실값: {epoch_valid_loss/len(loader_valid):.4f}')\n        print(f'\\t검증 데이터 ROC AUC : {roc_auc:.4f}') \n            \n        # if epoch_Loss is minimum of total epochs, model's parameter will be updated\n        # if epoch_valid_loss <= valid_loss_min:\n        #     print(f'\\t### 검증 데이터 손실값 업데이트 ({valid_loss_min/len(loader_valid):.4f} => {epoch_valid_loss/len(loader_valid):.4f})모델 저장')\n        \n        #     # minimum model parameter save\n        #     torch.save(model.state_dict(), save_parameter)\n        #     valid_loss_min = epoch_valid_loss\n        if roc_auc >= roc_auc_min:\n            print(f'\\t### Roc_Auc_Score 업데이트 ({roc_auc_min:.4f} => {roc_auc:.4f})모델 저장')\n            torch.save(model.state_dict(), save_parameter)\n            roc_auc_min = roc_auc\n\n        # 훈련 vs 검증 데이터 손실값 비교 그래프 용도 데이터 프레임 making (+ Roc_Auc Score)\n        loss_value = []\n        row = {'Train' : epoch_loss/len(loader_train), 'Validation' : epoch_valid_loss/len(loader_valid), 'Roc_Auc_Score' : roc_auc}\n        loss_value.append(row)\n        loss = pd.DataFrame(loss_value)\n        \n        df0 = pd.DataFrame(columns=column)\n        df0 = df0.merge(loss, how='outer', on=column, left_index=False,\n                    right_index=False)\n        df = df.append(df0, ignore_index=True).reset_index(drop=True)\n        print(df)\n        \n    return torch.load(save_parameter), df","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:49:37.168838Z","iopub.execute_input":"2022-09-28T02:49:37.16919Z","iopub.status.idle":"2022-09-28T02:49:37.185136Z","shell.execute_reply.started":"2022-09-28T02:49:37.169157Z","shell.execute_reply":"2022-09-28T02:49:37.184033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.8.4 Efficient-Net Train\n# 1st important value == Roc_Auc_Score\n# validation loss => it is important too.. but this value doesn't match exactly model's predict classification ability\nmodel_state_dict_efficient, df = train(criterion, optimizer1, model[0], loader_train, loader_valid, \n                                       epochs_efficient, scheduler1, save_parameter='efficient_model_state_dict.pth')","metadata":{"execution":{"iopub.status.busy":"2022-09-28T02:49:39.490065Z","iopub.execute_input":"2022-09-28T02:49:39.490745Z","iopub.status.idle":"2022-09-28T03:52:18.752374Z","shell.execute_reply.started":"2022-09-28T02:49:39.49071Z","shell.execute_reply":"2022-09-28T03:52:18.750907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.8.5 Train vs Validation Loss Visualization & Roc_Auc_Score => Efficient-Net\ndf['epochs_efficient'] = df.index + 1 \n\nsns.set_style(style=\"darkgrid\")\nplt.figure(figsize=(15,7))\n\ngraph = sns.lineplot(data=df, x='epochs_efficient', y='Train')\ngraph1 = sns.lineplot(data=df, x='epochs_efficient', y='Validation')\ngraph2 = sns.lineplot(data=df, x='epochs_efficient', y='Roc_Auc_Score')\n\nplt.legend(labels=['Train', 'Validation', 'Roc_Auc']) # 라벨 추가\ngraph.set(xlim=(0,epochs_efficient+1), xticks=np.arange(0,epochs_efficient+1,1)) # x축 간격 설정\ngraph.set(ylim=(0,1.0), yticks=np.arange(0,1,0.05))\ngraph.set(xlabel=\"Epoch\", ylabel=\"Loss Function Value & Roc_Auc\")\ngraph.set_title('Train, Vadlidation Loss History & Roc_Auc => Efficient-Net B05', fontsize=16)","metadata":{"execution":{"iopub.status.busy":"2022-09-13T07:07:38.915607Z","iopub.execute_input":"2022-09-13T07:07:38.916287Z","iopub.status.idle":"2022-09-13T07:07:39.364312Z","shell.execute_reply.started":"2022-09-13T07:07:38.916246Z","shell.execute_reply":"2022-09-13T07:07:39.363391Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.7 RegNet_Y 400mf\nmodel_state_dict_regent_y, df = train(criterion, optimizer2, model[1], loader_train, \n                                    loader_valid, epochs_regnet, scheduler2, save_parameter='regnet_y_model_state_dict.pth')","metadata":{"execution":{"iopub.status.busy":"2022-09-25T11:25:15.815277Z","iopub.execute_input":"2022-09-25T11:25:15.815651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3.8.5 Train vs Validation Loss Visualization => RegNet_Y 400mf\ndf['epochs_regnet'] = df.index + 1 \n\nsns.set_style(style=\"darkgrid\")\nplt.figure(figsize=(15,7))\n\ngraph = sns.lineplot(data=df, x='epochs_regnet', y='Train')\ngraph1 = sns.lineplot(data=df, x='epochs_regnet', y='Validation')\ngraph2 = sns.lineplot(data=df, x='epochs_regnet', y='Roc_Auc_Score')\n\nplt.legend(labels=['Train', 'Validation', 'Roc_Auc']) # 라벨 추가\ngraph.set(xlim=(0,epochs_regnet+1), xticks=np.arange(0,epochs_regnet+1,1)) # x축 간격 설정\ngraph.set(ylim=(0,1.0), yticks=np.arange(0,1,0.05))\ngraph.set(xlabel=\"Epoch\", ylabel=\"Loss Function Value & Roc_Auc\")\ngraph.set_title('Train, Vadlidation Loss History & Rog_Auc => RegNet_Y 400mf', fontsize=16)","metadata":{"execution":{"iopub.status.busy":"2022-09-12T15:45:14.883622Z","iopub.execute_input":"2022-09-12T15:45:14.884162Z","iopub.status.idle":"2022-09-12T15:45:15.478202Z","shell.execute_reply.started":"2022-09-12T15:45:14.88411Z","shell.execute_reply":"2022-09-12T15:45:15.47702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}