{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"0. Phì đại động mạch chủ\n1. Xẹp phổi\n2. Sự vôi hóa\n3. Tim to\n4. Củng cố\n5. ILD\n6. Sự xâm nhập\n7. Độ mờ của phổi\n8. Nodule / Mass\n9. Tổn thương khác\n10. Tràn dịch màng phổi\n11. Dày màng phổi\n12. Tràn khí màng phổi\n13. Xơ phổi\n14. \"Không tìm thấy\"","metadata":{}},{"cell_type":"code","source":"!pip install bbox_visualizer","metadata":{"_kg_hide-output":true,"execution":{"iopub.execute_input":"2022-05-21T02:07:43.8185Z","iopub.status.busy":"2022-05-21T02:07:43.81817Z","iopub.status.idle":"2022-05-21T02:07:52.849476Z","shell.execute_reply":"2022-05-21T02:07:52.848641Z","shell.execute_reply.started":"2022-05-21T02:07:43.818467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Thêm thư viện","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport os\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import model_selection\n\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport bbox_visualizer as bbv\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom glob import glob\nfrom skimage import exposure\n\nimport torch\n\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.utils.data.sampler import SequentialSampler\nimport torchvision\n\nfrom torchvision.models.detection.faster_rcnn import FastRCNNPredictor\nfrom torchvision.models.detection import FasterRCNN\n\nimport albumentations as A\nfrom albumentations.pytorch.transforms import ToTensorV2\n\nimport warnings\n\nwarnings.filterwarnings('ignore')","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2022-05-21T02:07:55.484538Z","iopub.status.busy":"2022-05-21T02:07:55.484195Z","iopub.status.idle":"2022-05-21T02:07:57.714883Z","shell.execute_reply":"2022-05-21T02:07:57.714064Z","shell.execute_reply.started":"2022-05-21T02:07:55.484504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = '../input/vinbigdata-chest-xray-abnormalities-detection/'\nWEIGHTS_FILE = './'","metadata":{"execution":{"iopub.execute_input":"2022-05-21T02:09:38.382991Z","iopub.status.busy":"2022-05-21T02:09:38.382578Z","iopub.status.idle":"2022-05-21T02:09:38.387438Z","shell.execute_reply":"2022-05-21T02:09:38.386494Z","shell.execute_reply.started":"2022-05-21T02:09:38.382953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = pd.read_csv(os.path.join(BASE_DIR, \"train.csv\")) # đưa dữ liệu vào pandas\ndataset.head()","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.execute_input":"2022-05-21T02:09:41.274711Z","iopub.status.busy":"2022-05-21T02:09:41.274386Z","iopub.status.idle":"2022-05-21T02:09:41.418769Z","shell.execute_reply":"2022-05-21T02:09:41.417889Z","shell.execute_reply.started":"2022-05-21T02:09:41.274679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# vẽ biểu đồ\nplt.figure(figsize=(16, 9))\n\nsns.countplot(dataset.class_name)\nplt.xticks(fontsize=14,rotation=65)\nplt.yticks(fontsize=14)\nplt.title(\"Phân phối nhãn trong khung dữ liệu chú thích\", fontsize=16)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T02:14:15.306581Z","iopub.status.busy":"2022-05-21T02:14:15.306249Z","iopub.status.idle":"2022-05-21T02:14:15.536386Z","shell.execute_reply":"2022-05-21T02:14:15.53555Z","shell.execute_reply.started":"2022-05-21T02:14:15.306549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_new = dataset[dataset.class_name!='No finding'].reset_index(drop=True) # bỏ đi nhãn no finding và reset lại vị trí\n\n# vẽ biểu đồ sau khi loại bỏ no finding\nplt.figure(figsize=(16, 9))\n\nsns.countplot(dataset_new.class_name)\nplt.xticks(fontsize=14,rotation=65)\nplt.yticks(fontsize=14)\nplt.title(\"Phân phối nhãn trong khung dữ liệu chú thích\", fontsize=16);","metadata":{"execution":{"iopub.execute_input":"2022-05-21T02:16:21.332211Z","iopub.status.busy":"2022-05-21T02:16:21.331831Z","iopub.status.idle":"2022-05-21T02:16:21.579871Z","shell.execute_reply":"2022-05-21T02:16:21.578951Z","shell.execute_reply.started":"2022-05-21T02:16:21.332175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, valid_df = train_test_split(dataset_new, test_size=0.33, random_state=42) # chia dữ liệu","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:44:47.58633Z","iopub.status.busy":"2022-05-20T15:44:47.585754Z","iopub.status.idle":"2022-05-20T15:44:47.598717Z","shell.execute_reply":"2022-05-20T15:44:47.59772Z","shell.execute_reply.started":"2022-05-20T15:44:47.586129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Thiết kế Dataset Loader","metadata":{}},{"cell_type":"code","source":"# Xây dựng Tập dữ liệu chú thích phổi\nclass LungsAnnotationDataset(Dataset):\n    def __init__(self,dataframe, image_dir, transforms=None):\n        super().__init__()\n        \n        self.image_ids = dataframe['image_id'].unique()\n        self.df = dataframe\n        self.image_dir = image_dir\n        self.transforms = transforms\n    \n    def __getitem__(self, index: int):\n        image_id = self.image_ids[index]\n        records = self.df[self.df['image_id'] == image_id]\n        # đọc file dicom\n        dcm_data = pydicom.read_file(f'{self.image_dir}/{image_id}.dicom')\n        image = apply_voi_lut(dcm_data.pixel_array, dcm_data)\n        # tùy thuộc vào giá trị này, tia X có thể bị đảo ngược\n        if dcm_data.PhotometricInterpretation == \"MONOCHROME1\":\n            image = np.amax(image) - image\n            \n        image = np.stack([image, image, image])\n        image = image - np.min(image)\n\n        image = image / image.max()\n\n        image = exposure.equalize_hist(image) # Chuẩn hóa hình ảnh tia X\n        image = image.astype('float32')\n\n        image = image.transpose(1,2,0)\n        \n        boxes = records[['x_min','y_min','x_max','y_max']].values #boxes[:, 2] = boxes[:, 0] + boxes[:, 2]\n        \n        area = (boxes[:, 3] - boxes[:, 1]) * (boxes[:, 2] - boxes[:, 0])\n        area = torch.as_tensor(area, dtype=torch.float32)\n        \n        labels = records.class_id.values + 1\n        class_name = records.class_name.values\n        #giả sử tất cả các trường hợp không phải là đa số\n        iscrowd = torch.zeros((records.shape[0],), dtype=torch.int64)\n        \n        target = {}\n        target['boxes'] = torch.tensor(boxes)\n        target['labels'] = torch.tensor(labels)\n        target['area'] = torch.tensor(area)\n        target['iscrowd'] = torch.tensor(iscrowd)\n        \n        if self.transforms:\n            sample = {\n                'image': image,\n                'bboxes': target['boxes'],\n                'labels': labels\n            }\n            sample = self.transforms(**sample)\n            image = sample['image']\n            \n            target['boxes'] = torch.tensor(sample['bboxes'])\n            \n        return image, target\n    \n    def __len__(self) -> int:\n        return self.image_ids.shape[0]","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:44:47.601714Z","iopub.status.busy":"2022-05-20T15:44:47.601366Z","iopub.status.idle":"2022-05-20T15:44:47.618947Z","shell.execute_reply":"2022-05-20T15:44:47.617958Z","shell.execute_reply.started":"2022-05-20T15:44:47.601687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Albumentations\n# ToTensorV2: Chuyển đổi hình ảnh và mặt nạ thành` torch.Tensor`. \n# Hình ảnh `HWC` numpy được chuyển đổi thành pytorch` CHW` tensor.\ndef get_train_transform():\n    return A.Compose([ToTensorV2(),], bbox_params={'format': 'pascal_voc', 'label_fields': ['labels']})\n\ndef get_valid_transform():\n    return A.Compose([ToTensorV2(),], bbox_params={'format': 'pascal_voc', 'label_fields': ['labels']})","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:44:47.621411Z","iopub.status.busy":"2022-05-20T15:44:47.620955Z","iopub.status.idle":"2022-05-20T15:44:47.630911Z","shell.execute_reply":"2022-05-20T15:44:47.630058Z","shell.execute_reply.started":"2022-05-20T15:44:47.621371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DIR_TRAIN = os.path.join(BASE_DIR, \"train\")\n# gắn nhãn cho bộ dữ liệu\ntrain_dataset = LungsAnnotationDataset(train_df, DIR_TRAIN,get_train_transform())\nvalid_dataset = LungsAnnotationDataset(valid_df, DIR_TRAIN,get_valid_transform())","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:44:47.633898Z","iopub.status.busy":"2022-05-20T15:44:47.633244Z","iopub.status.idle":"2022-05-20T15:44:47.645995Z","shell.execute_reply":"2022-05-20T15:44:47.645285Z","shell.execute_reply.started":"2022-05-20T15:44:47.633858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def collate_fn(batch):\n    return tuple(zip(*batch))\n\ntrain_data_loader = DataLoader(\n    train_dataset,\n    batch_size=6,\n    shuffle=False,\n    num_workers=4,\n    collate_fn=collate_fn\n) # xử lý 1 loạt 6 ảnh và 6 nhãn liên tiếp theo batch_size\n\nvalid_data_loader = DataLoader(\n    valid_dataset,\n    batch_size=6,\n    shuffle=False,\n    num_workers=4,\n    collate_fn=collate_fn\n) # xử lý 1 loạt 6 ảnh và 6 nhãn liên tiếp theo batch_size","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:44:47.648356Z","iopub.status.busy":"2022-05-20T15:44:47.647715Z","iopub.status.idle":"2022-05-20T15:44:47.653996Z","shell.execute_reply":"2022-05-20T15:44:47.652886Z","shell.execute_reply.started":"2022-05-20T15:44:47.648319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# device = torch.device('cuda') \nif torch.cuda.is_available():\n    device = torch.device('cuda')\nelse:\n    device = torch.device('cpu')","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:44:47.655922Z","iopub.status.busy":"2022-05-20T15:44:47.655665Z","iopub.status.idle":"2022-05-20T15:44:47.717677Z","shell.execute_reply":"2022-05-20T15:44:47.716788Z","shell.execute_reply.started":"2022-05-20T15:44:47.655898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#iter() gọi __iter__()phương thức iris_loader trả về một trình lặp. \n#next() sau đó gọi __next__()phương thức trên trình lặp đó để nhận lần lặp đầu tiên. \n#Chạy next()lại sẽ nhận được mục thứ hai của trình lặp, v.v\nimages, targets = next(iter(train_data_loader))\nimages = list(image.to(device) for image in images)\n\ntargets = [{k: v.to(device) for k, v in t.items()} for t in targets]","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:44:47.719514Z","iopub.status.busy":"2022-05-20T15:44:47.719096Z","iopub.status.idle":"2022-05-20T15:45:47.589243Z","shell.execute_reply":"2022-05-20T15:45:47.587888Z","shell.execute_reply.started":"2022-05-20T15:44:47.719471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_brands = {\n    0: 'Aortic enlargement',\n    1: 'Atelectasis',\n    2: 'Calcification',\n    3: 'Cardiomegaly',\n    4: 'Consolidation',\n    5: 'ILD',\n    6: 'Infiltration',\n    7: 'Lung Opacity',\n    8: 'Nodule/Mass',\n    9: 'Other lesion',\n    10: 'Pleural effusion',\n    11: 'Pleural thickening',\n    12: 'Pneumothorax',\n    13: 'Pulmonary fibrosis'\n}","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Trực quan hóa các bất thường trên X-quang ngực","metadata":{}},{"cell_type":"code","source":"def plot_x_ray(idx,images,targets):\n    boxes = targets[idx]['boxes'].cpu().numpy().astype(np.int32)\n    labels = targets[idx]['labels']-1\n    sample = images[idx].permute(1,2,0).cpu().numpy()\n    \n    img = sample.copy()\n    plt.figure(figsize=(16, 16))\n    for box,label in zip(boxes,labels):\n        bbv.add_label(img, class_brands[label.item()], box, \n                      draw_bg=True,\n                      text_bg_color=(255,0,0),\n                      text_color=(0,0,0),\n                      )\n        cv2.rectangle(img ,\n                      (box[0], box[1]),\n                      (box[2], box[3]),\n                      (255,0,0), 3)\n\n    plt.imshow(img)    ","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:45:47.591217Z","iopub.status.busy":"2022-05-20T15:45:47.590815Z","iopub.status.idle":"2022-05-20T15:45:47.603767Z","shell.execute_reply":"2022-05-20T15:45:47.602909Z","shell.execute_reply.started":"2022-05-20T15:45:47.591178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_x_ray(2,images,targets) # hiển thị các loại bất thường kèm vị trí","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:45:47.605911Z","iopub.status.busy":"2022-05-20T15:45:47.605249Z","iopub.status.idle":"2022-05-20T15:45:49.076848Z","shell.execute_reply":"2022-05-20T15:45:49.075868Z","shell.execute_reply.started":"2022-05-20T15:45:47.605866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_x_ray(3,images,targets)","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:45:49.07832Z","iopub.status.busy":"2022-05-20T15:45:49.077994Z","iopub.status.idle":"2022-05-20T15:45:50.497815Z","shell.execute_reply":"2022-05-20T15:45:50.496982Z","shell.execute_reply.started":"2022-05-20T15:45:49.078287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_x_ray(1,images,targets)","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:45:50.499886Z","iopub.status.busy":"2022-05-20T15:45:50.49934Z","iopub.status.idle":"2022-05-20T15:45:52.552926Z","shell.execute_reply":"2022-05-20T15:45:52.552078Z","shell.execute_reply.started":"2022-05-20T15:45:50.499845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Xây dựng Mô hình","metadata":{}},{"cell_type":"code","source":"# Thay thế classifier bằng một trình phân loại mới, có num_classes do người dùng xác định\nnum_classes = 15  # 14 lớp + background\n\n# tải một mô hình; được đào tạo trước về COCO\nmodel = torchvision.models.detection.fasterrcnn_resnet50_fpn(pretrained=True)\n# nhận số lượng các đặc trưng đầu vào cho classifier\nin_features = model.roi_heads.box_predictor.cls_score.in_features\n\n# thay thế đầu được đào tạo trước bằng một đầu mới của FastRcnn\nmodel.roi_heads.box_predictor = FastRCNNPredictor(in_features, num_classes)","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:45:52.554923Z","iopub.status.busy":"2022-05-20T15:45:52.554327Z","iopub.status.idle":"2022-05-20T15:45:59.224832Z","shell.execute_reply":"2022-05-20T15:45:59.223898Z","shell.execute_reply.started":"2022-05-20T15:45:52.554858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.to(device)\nparams = [p for p in model.parameters() if p.requires_grad]\noptimizer = torch.optim.SGD(params, lr=0.0005, momentum=0.9, weight_decay=0.0005) #khởi tạo optimizer.\nlr_scheduler = None\n\n# chọn sô epoch = 4\nnum_epochs = 4","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:45:59.22724Z","iopub.status.busy":"2022-05-20T15:45:59.226779Z","iopub.status.idle":"2022-05-20T15:45:59.29597Z","shell.execute_reply":"2022-05-20T15:45:59.295234Z","shell.execute_reply.started":"2022-05-20T15:45:59.22718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hàm tính loss\nclass Averager:\n    def __init__(self):\n        self.current_total = 0.0\n        self.iterations = 0.0\n\n    def send(self, value):\n        self.current_total += value\n        self.iterations += 1\n\n    @property\n    def value(self):\n        if self.iterations == 0:\n            return 0\n        else:\n            return 1.0 * self.current_total / self.iterations\n\n    def reset(self):\n        self.current_total = 0.0\n        self.iterations = 0.0","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:45:59.298071Z","iopub.status.busy":"2022-05-20T15:45:59.297577Z","iopub.status.idle":"2022-05-20T15:45:59.305512Z","shell.execute_reply":"2022-05-20T15:45:59.304492Z","shell.execute_reply.started":"2022-05-20T15:45:59.298032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_validation(idx, images, targets, predicted):\n    boxes = targets[idx]['boxes'].cpu().numpy().astype(np.int32)\n    labels = targets[idx]['labels']-1\n    \n    boxes_pred = predicted[idx]['boxes']#.detach().numpy().astype(np.int32)\n    labels_pred = predicted[idx]['labels']-1\n    scores = predicted[idx]['scores'].data.cpu().numpy()\n    # Threshold: ngưỡng chấp nhận\n    boxes_pred = boxes_pred[scores >= 0.2]\n    \n    sample = images[idx].permute(1,2,0).cpu().numpy()\n    \n    img = sample.copy() # tạo 1 mẫu bản sao\n    plt.figure(figsize=(16, 16))\n    \n    for box,label in zip(boxes,labels):\n        bbv.add_label(img, class_brands[label.item()], box, \n                      draw_bg=True,\n                      text_bg_color=(255,0,0),\n                      text_color=(0,0,0),\n                      )\n        cv2.rectangle(img ,\n                      (box[0], box[1]),\n                      (box[2], box[3]),\n                      (255,0,0), 3)\n    for box,label in zip(boxes_pred,labels_pred):\n        bbv.add_label(img, class_brands[label.item()], box, \n                      draw_bg=True,\n                      text_bg_color=(255, 255, 0), #đỏ\n                      text_color=(0,0,0),\n                      )\n        cv2.rectangle(img ,\n                      (box[0], box[1]),\n                      (box[2], box[3]),\n                      (255, 255, 0), 3) # vàng\n\n    plt.imshow(img)\n    plt.title('Vùng tin tưởng - Đỏ, Hộp đã dự đoán - Vàng',fontsize=14)","metadata":{"execution":{"iopub.execute_input":"2022-05-20T15:45:59.30792Z","iopub.status.busy":"2022-05-20T15:45:59.307471Z","iopub.status.idle":"2022-05-20T15:45:59.326469Z","shell.execute_reply":"2022-05-20T15:45:59.325483Z","shell.execute_reply.started":"2022-05-20T15:45:59.307879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(train_dataset, fold):\n    train_data_loader = DataLoader(\n        train_dataset,\n        batch_size=2,\n        shuffle=False,\n        num_workers=4,\n        collate_fn=collate_fn\n    )\n    loss_hist = Averager() # khởi tạo 1 đối tượng tính trung bình\n    itr = 1\n    for epoch in range(num_epochs):\n        loss_hist.reset() #lấy lại giá trị gốc\n        model.train()\n        for images, targets in train_data_loader:\n            optimizer.zero_grad()\n            images = list(image.to(device) for image in images)\n            targets = [{k: v.to(device) for k, v in t.items()} for t in targets]\n\n            loss_dict = model(images, targets)\n\n            loss_classifier, loss_box_reg, loss_objectness, loss_rpn_box_reg = loss_dict.values()\n            losses = sum([loss_objectness, \n                         10*loss_classifier, \n                         10*loss_rpn_box_reg, \n                         (0.5*loss_box_reg**2)\n                         ]\n                        )\n            loss_value = losses.item()\n            loss_hist.send(loss_value)\n\n            losses.backward()\n            optimizer.step()\n            if itr%100==0:\n                print(f\"Fold #{fold} Epoch #{epoch+1} Iteration #{itr} loss: {loss_hist.value}\")\n            \n            itr += 1\n            del loss_dict, loss_classifier, loss_box_reg, loss_objectness, loss_rpn_box_reg,loss_value\n        itr=1    \n        # cập nhật tốc độ học\n        if lr_scheduler is not None:\n            lr_scheduler.step()\n\n        print(f\"Fold #{fold} Epoch #{epoch+1} loss: {loss_hist.value}\")\n        print(\"Saving Epoch's state...\")\n        torch.save(model.state_dict(), f\"model_fasterRCNN.pth\") #lưu model để tái sử dụng","metadata":{"execution":{"iopub.status.busy":"2022-05-22T05:15:52.642134Z","iopub.execute_input":"2022-05-22T05:15:52.642511Z","iopub.status.idle":"2022-05-22T05:15:52.660936Z","shell.execute_reply.started":"2022-05-22T05:15:52.642479Z","shell.execute_reply":"2022-05-22T05:15:52.659532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Huấn luyện cho mô hình","metadata":{}},{"cell_type":"code","source":"k = 1\ndf = dataset_new.sample(frac=1).reset_index(drop=True)\ny = dataset_new.class_id.values\n## Chia tỉ lệ bằng K-fold\nkfold = model_selection.GroupKFold(n_splits=5)\n\nfor train_index, val_index in kfold.split(df, y,groups=df.image_id.values):\n    train_dataset = LungsAnnotationDataset(df.loc[val_index], DIR_TRAIN,get_train_transform())\n    train_model(train_dataset, k)\n    if k==5:\n        valid_dataset = LungsAnnotationDataset(df.loc[train_index], DIR_TRAIN,get_train_transform())\n        val_data_loader = DataLoader(valid_dataset,\n                                batch_size=6,\n                                shuffle=False,\n                                num_workers=4,\n                                collate_fn=collate_fn\n                            ) # tải dữ liệu valid vs bó là 6\n        \n    k += 1","metadata":{"_kg_hide-output":false,"execution":{"iopub.execute_input":"2022-05-20T15:45:59.346147Z","iopub.status.busy":"2022-05-20T15:45:59.345499Z","iopub.status.idle":"2022-05-21T00:38:10.095211Z","shell.execute_reply":"2022-05-21T00:38:10.092731Z","shell.execute_reply.started":"2022-05-20T15:45:59.346103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iter_ = iter(val_data_loader) #lặp\nimages_val0, targets_val0 = next(iter_) # nhận lần lặp đầu tiên\nimages_val1, targets_val1 = next(iter_) # nhận lần lặp thứ 2","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:38:10.101Z","iopub.status.busy":"2022-05-21T00:38:10.100256Z","iopub.status.idle":"2022-05-21T00:38:58.600687Z","shell.execute_reply":"2022-05-21T00:38:58.597661Z","shell.execute_reply.started":"2022-05-21T00:38:10.100938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load lại mô hình","metadata":{}},{"cell_type":"code","source":"model.load_state_dict(torch.load(WEIGHTS_FILE+'model_fasterRCNN.pth', map_location=device)) # load lại model đã save ở trên\nmodel.eval() #tắt các lớp như Dropouts Layers, BatchNorm Layers, v.v trong khi đánh giá mô hình\n\nmodel.to(device) # chỉ cần chuyển đổi mô hình đã khởi tạo thành cpu or cuda tùy thuộc tham số trên\nmodel.eval()\ndevice = torch.device(\"cuda\")\nimages_0 = list(image.to(device) for image in images_val0)\noutputs0 = model(images_0)\noutputs0 = [{k: v.to(device) for k, v in t.items()} for t in outputs0]","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:38:58.615393Z","iopub.status.busy":"2022-05-21T00:38:58.609851Z","iopub.status.idle":"2022-05-21T00:39:00.87688Z","shell.execute_reply":"2022-05-21T00:39:00.873782Z","shell.execute_reply.started":"2022-05-21T00:38:58.615297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_1 = list(image_.to(device) for image_ in images_val1)\noutputs1 = model(images_1)\noutputs1 = [{k: v.to(device) for k, v in t.items()} for t in outputs1]","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:39:00.892303Z","iopub.status.busy":"2022-05-21T00:39:00.88663Z","iopub.status.idle":"2022-05-21T00:39:01.938777Z","shell.execute_reply":"2022-05-21T00:39:01.936049Z","shell.execute_reply.started":"2022-05-21T00:39:00.892247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Trực quan hóa hiệu suất của mô hình qua hình ảnh validation","metadata":{}},{"cell_type":"code","source":"plot_validation(0, images_val0, targets_val0, outputs0)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:39:01.940996Z","iopub.status.busy":"2022-05-21T00:39:01.940615Z","iopub.status.idle":"2022-05-21T00:39:10.969746Z","shell.execute_reply":"2022-05-21T00:39:10.968185Z","shell.execute_reply.started":"2022-05-21T00:39:01.940955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_validation(0, images_val1, targets_val1, outputs1)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:39:10.980606Z","iopub.status.busy":"2022-05-21T00:39:10.974106Z","iopub.status.idle":"2022-05-21T00:39:16.549255Z","shell.execute_reply":"2022-05-21T00:39:16.539218Z","shell.execute_reply.started":"2022-05-21T00:39:10.980542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_validation(1, images_val0, targets_val0, outputs0)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:39:16.558417Z","iopub.status.busy":"2022-05-21T00:39:16.552468Z","iopub.status.idle":"2022-05-21T00:39:25.783127Z","shell.execute_reply":"2022-05-21T00:39:25.778992Z","shell.execute_reply.started":"2022-05-21T00:39:16.558314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_validation(1, images_val1, targets_val1, outputs1)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:39:25.805003Z","iopub.status.busy":"2022-05-21T00:39:25.804503Z","iopub.status.idle":"2022-05-21T00:39:32.748951Z","shell.execute_reply":"2022-05-21T00:39:32.747751Z","shell.execute_reply.started":"2022-05-21T00:39:25.804955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_validation(2, images_val0, targets_val0, outputs0)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:39:32.752783Z","iopub.status.busy":"2022-05-21T00:39:32.752352Z","iopub.status.idle":"2022-05-21T00:39:40.178867Z","shell.execute_reply":"2022-05-21T00:39:40.177672Z","shell.execute_reply.started":"2022-05-21T00:39:32.752736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_validation(2, images_val1, targets_val1, outputs1)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:39:40.182489Z","iopub.status.busy":"2022-05-21T00:39:40.181444Z","iopub.status.idle":"2022-05-21T00:39:45.326347Z","shell.execute_reply":"2022-05-21T00:39:45.324812Z","shell.execute_reply.started":"2022-05-21T00:39:40.182416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_validation(3, images_val0, targets_val0, outputs0)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:39:45.335306Z","iopub.status.busy":"2022-05-21T00:39:45.328615Z","iopub.status.idle":"2022-05-21T00:39:49.001535Z","shell.execute_reply":"2022-05-21T00:39:49.000142Z","shell.execute_reply.started":"2022-05-21T00:39:45.335231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_validation(3, images_val1, targets_val1, outputs1)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:39:49.008203Z","iopub.status.busy":"2022-05-21T00:39:49.003516Z","iopub.status.idle":"2022-05-21T00:39:53.797591Z","shell.execute_reply":"2022-05-21T00:39:53.796214Z","shell.execute_reply.started":"2022-05-21T00:39:49.00815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_validation(4, images_val0, targets_val0, outputs0)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:39:53.800279Z","iopub.status.busy":"2022-05-21T00:39:53.799503Z","iopub.status.idle":"2022-05-21T00:39:59.168855Z","shell.execute_reply":"2022-05-21T00:39:59.167418Z","shell.execute_reply.started":"2022-05-21T00:39:53.800228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_validation(4, images_val1, targets_val1, outputs1)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:39:59.17196Z","iopub.status.busy":"2022-05-21T00:39:59.17137Z","iopub.status.idle":"2022-05-21T00:40:03.240032Z","shell.execute_reply":"2022-05-21T00:40:03.238707Z","shell.execute_reply.started":"2022-05-21T00:39:59.171901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_validation(5, images_val0, targets_val0, outputs0)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:40:03.242542Z","iopub.status.busy":"2022-05-21T00:40:03.242005Z","iopub.status.idle":"2022-05-21T00:40:07.018718Z","shell.execute_reply":"2022-05-21T00:40:07.017193Z","shell.execute_reply.started":"2022-05-21T00:40:03.242494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_validation(5, images_val1, targets_val1, outputs1)","metadata":{"execution":{"iopub.execute_input":"2022-05-21T00:40:07.024203Z","iopub.status.busy":"2022-05-21T00:40:07.020996Z","iopub.status.idle":"2022-05-21T00:40:11.216264Z","shell.execute_reply":"2022-05-21T00:40:11.215057Z","shell.execute_reply.started":"2022-05-21T00:40:07.02415Z"},"trusted":true},"execution_count":null,"outputs":[]}]}