{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train.csv')\ntrain","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train.fillna(0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.loc[train['class_name'] == 'No finding',['x_max', 'y_max']] = 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# FastRCNN 계열은 배경 (no finding) 의 정답값은 무조건 0 이되어야함.\n# 여기 class_id 에선 14 번으로 되어있음 (NaN 값들 보면알수있음)"},{"metadata":{"trusted":true},"cell_type":"code","source":"train['class_id'] = train['class_id'] +1\n\ntrain.loc[train['class_id'] == 15,'class_id'] = 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[train['image_id'] == '9a5094b2563a1ef3ff50dc5c7ff71345']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[train['image_id'] == 'ca7e72954550eeb610fe22bf0244b7fa']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from torch.utils.data import Dataset, DataLoader\nimport pydicom\n\nclass VinBigDataset(Dataset):\n    def __init__(self,dataframe, image_dir, transforms = None):\n        super().__init__()\n        self.image_ids = dataframe['image_id'].unique()\n        self.dataframe = dataframe\n        self.image_dir = image_dir\n        self.transforms = transforms\n        \n    def __getitem__(self, index):\n        image_id = self.image_ids[index]\n        # records --> 직접 학습을 할때, bbox 들을 관리하고 학습을 할수있게 도와줌.\n        # 아래는 index 가 정렬이 안되어있음 -> reset_index\n        records = self.dataframe[self.dataframe['image_id'] == image_id].reset_index(drop = True)\n        dicom = pydicom.dcmread(self.image_dir + image_id + '.dicom')\n        image = dicom.pixel_array\n        # dicom 파일은 사람들이 움직이거나 하면 y 절편이 움직임 or 기울기가 변하게됌 --> 밝기에 영향을 줌 \n        \n        intercept = dicom.RescaleIntercept if 'RescaleIntercept' in dicom else 0\n        slope = dicom.RescaleSlope if 'RescaleSlope' in dicom else 1\n        \n        if slope != 1:\n            image = slope * image\n            image = image.astype(np.int16)\n        \n        image += np.int16(intercept)\n        \n        # 3채널로 변경, float32 는 메모리 줄이기위해 쓰는것\n        image = np.stack([image, image, image]).astype(np.float32)\n        # 표준화, dicom 파일은 경사 (y절편이) 제대로 안되어있으면 (ex) 사람이 움직이면) 사진의 밝기에 영향을줌 --> pixel 값의 최소값이 0 이아닐수도있음\n        image = image - image.min()\n        # 최대값으로 나눠서 0과1사이로 변환\n        image = image / image.max()\n        # 채널값을 뒤로 옮김 원랜 0,1,2 순\n        image = image.transpose(1,2,0)\n        \n        # 특정 사진을 가져왔을때, 그게 만약에 0 번이라면 (no finding), 정상인 애들\n        if records.loc[0, 'class_id'] == 0:\n            # 3개를 가져오니깐 하나만 가져오게 하는코드\n            records = records.loc[[0],:]\n            \n        boxes = records[['x_min', 'y_min', 'x_max', 'y_max']].values\n        \n        # 각각의 바운딩박스마다 정답값(label) 을 넣어주는것 \n        labels = torch.tensor(records['class_id'].values,dtype = torch.int64)\n        # 항상 key 값은 아래와 동일 (boxes, labels)\n        target = {'boxes' : boxes, 'labels' : labels}\n        if self.transforms:\n            # 딕셔너리 target 은 위에 만드는데 왜 또 딕셔너리 sample 을 만드는지? :\n            # target 은 학습하거나 예측할때 쓰이는 정답값 (bounding box 와 labels)\n            # sample 은 augmentation (albumentation) 할때 쓰이는 정답값들. 항상 key 값은 아래와 동일\n            sample = {'image' : image, 'bboxes' : boxes, 'labels' : labels}\n            sample = self.transforms(**sample)\n            # 딕셔너리 키값에 접근하는법\n            image = sample['image']\n            # 텐서형식으로 바뀐 애들 저장\n            # boxes 를 위에서 미리 변경을 안해주고 augmentation 돌고나서 해주는이유 : 이미지가 뒤집히면서 bounding box 도 뒤집혀야하기때문\n            target['boxes'] = torch.tensor(sample['bboxes'])\n            # tensor 형식으로 이미 위에서 바꿨기때문에 필요없음 --> target['labels'] = sample['labels']\n        return image, target\n        \n    \n    def __len__(self):\n        return len(self.image_ids)\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from albumentations.pytorch.transforms import ToTensorV2\n\nimport albumentations as al","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 이미지 사이즈 동일하게 변경해줘야함\ndef get_train_transform():\n    return al.Compose([al.Resize(height = 512, width = 512), al.Flip(0.5),ToTensorV2()], bbox_params ={'format' : 'pascal_voc', 'label_fields': ['labels']})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_valid_transform():\n    return al.Compose([al.Resize(height = 512, width = 512),ToTensorV2()], bbox_params ={'format' : 'pascal_voc', 'label_fields': ['labels']})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_dataset = VinBigDataset(train, '/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/', get_train_transform())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 나중에는 train 이 아니라 valid 가 들어감. train_test_split\n\nvalid_dataset = VinBigDataset(train, '/kaggle/input/vinbigdata-chest-xray-abnormalities-detection/train/', get_valid_transform())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def collate_fn(batch):\n    return tuple(zip(*batch))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_dataloader = DataLoader(train_dataset, batch_size = 8, num_workers = 4, collate_fn = collate_fn)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid_dataloader = DataLoader(valid_dataset, batch_size = 8, num_workers = 4, collate_fn = collate_fn)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train\n\n# wheat dataset 은 이미지 사이즈가 다 같음. 이건 다 다름.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torchvision","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from torchvision.models.detection.faster_rcnn import FastRCNNPredictor\n\nmodel = torchvision.models.detection.fasterrcnn_resnet50_fpn(pretrained=True)\n\n# 맨마지막 출력층을 클래스 갯수로 (91개 --> 15개)\n# 맨 마지막 줄도 15 * 4 --> 60 으로 바뀜\n\nmodel.roi_heads.box_predictor = FastRCNNPredictor(1024, 15)\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import torch\n\ndevice = torch.device('cuda')\n\nmodel.to(device)\n\n# requires_grad --> 학습을 할수 있는 층, 없는 층이 있다. true 인 애들만 가져올것. freeze 되어있는애들 제외.\n\nparams = [x for x in model.parameters() if x.requires_grad]\noptimizer = torch.optim.Adam(params)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# 모델마다 달라짐. FasterRCNN 에서 학습할때 Loss 를 어디에 저장하고, 어떻게 업데이트 할건지 정하는것.\n\nclass Averager:\n    def __init__(self):\n        self.current_total = 0.0\n        self.iterations = 0.0\n\n    def send(self, value):\n        self.current_total += value\n        self.iterations += 1\n\n    @property\n    def value(self):\n        if self.iterations == 0:\n            return 0\n        else:\n            # Loss 에 대한 평균을 구한다는것. 1.0 곱하는것은 loss 가 float 으로 나와야 오류가 안나서\n            return 1.0 * self.current_total / self.iterations\n\n    def reset(self):\n        # 한번 학습을 했으면 새로운 epoch 이 시작되기전에 초기화\n        self.current_total = 0.0\n        self.iterations = 0.0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import warnings\n\nwarnings.filterwarnings('ignore')\n\nnum_epochs = 2\niteration = 1\n\nloss_hist = Averager()\n\n# 학습시작\n\nfor epoch in range(num_epochs):\n    # epoch 마다 초기화해야함\n    loss_hist.reset()\n    for images, targets in train_dataloader:\n        images = [x.to(device) for x in images]\n        # 오른쪽의 targets 은 8개. t 는 그중 하나. 왼쪽 for 문은 그 하나의 딕셔너리를 접근했을때 key 값과 value 값을 접근해야함\n        targets = [{k: v.to(device) for k,v in t.items()} for t in targets]\n        # 분류문제와는 달리 detection 은 여러개의 오류값이나옴\n        loss_dict = model(images, targets)\n        # 딕셔너리에 있는 values 함수이기 때문에 () 가 필요\n        losses = sum(x for x in loss_dict.values())\n        # 텐서처럼 [] 로 감싸고 있는데 .item 해서 숫자로만 가져옴\n        loss_value = losses.item()\n        loss_hist.send(loss_value)\n        # 가중치 초기화, 항상 이 3줄이나옴\n        optimizer.zero_grad()\n        # backpropagation\n        losses.backward()\n        # w 값에 lr 을 곱해줘서 update 를 해줌\n        optimizer.step()\n        # verbose 옵션. loss 값도 봐야함\n        if iteration % 1 == 0:\n            print(f'Iteration :{iteration} , Loss : {loss_value}')\n            \n        iteration += 1\n    print(f'Epoch : {epoch}, Loss : {loss_hist.value}')\n    \n    \n\ntorch.save(model.state_dict(), 'best.pth')\n        \n        \n\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Loss 값이 왔다갔다 심함 --> keras 와 다름\n# Keras 도 원래 이렇게 나옴. 이런애들을 바로안보내고 옛날에 있던 loss 들이 계속 저장이 되면서 평균값으로 나옴. \n"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Valid 데이터셋을 가져와서 detection 잘맞는지 틀리는지 그림그려보기, 리더보드 제출안하고 하는것\n\n# iter --> 데이터를 8장씩 가져옴\n# next --> 이번거 가져옴\n\nimages , targets = next(iter(valid_dataloader))\n\nimages = [image.to(device) for image in images]\n\ntargets = [{k: v.to(device) for k, v in t.items()} for t in targets]\n\nboxes = targets[0]['boxes'].cpu().numpy()\n\n# image 는 targets 와 다름. 그냥 image\nsample = images[0].permute(1,2,0).cpu().numpy()\n\n#예측하기\n\nmodel.eval()\n\noutputs = model(images)\n\ncpu_device = torch.device(\"cpu\")\noutputs = [{k: v.to(cpu_device) for k, v in t.items()} for t in outputs]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport cv2\n\na, b = plt.subplots(1,1, figsize = (20,12))\n\nfor i in boxes:\n    # (xmin, ymin)  , (xmax, ymax)\n    cv2.rectangle(sample, (i[0], i[1]), (i[2], i[3]), (100,0,0), 2)\n\nb.set_axis_off()\nb.imshow(sample)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}