{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Let's use artificial dataset\n\nRecently I made some artificial data. I simply did copy the disease bbox area and paste to normal image ( which was originally tagged 14, normal class ). and made new annotation for pasted bbox coordinates. You can easily find my dataset [link](https://www.kaggle.com/seokhyunseo/artificial-vinbigdatacoco-format). Please upvote :). I will upload the dataset making codes soon, after I did some refactoring.\n\n# Result  \n\nThe baseline result is from [here](https://www.kaggle.com/seokhyunseo/let-s-use-mmdetections-different-models)(not closed notebook, I will upload more contents there and don't forget upvote). I use same validation set for checking\n\n| model | 5epoch | 10epoch | \n|---:|---:|---:|\n|CascadeRCNN(Resnet50)|0.182|0.239|\n|CascadeRCNN(New dataset) | 0.201 | 0.269 |\n\nI found the more dataset leads to more performance. The quality of artificial images will lead more performance.\n"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"!pip -qq install mmcv-full","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!git clone https://github.com/open-mmlab/mmdetection.git\n    \n%cd mmdetection\n\n!pip -qq install -e .","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Pretrain Model Download\n\n!mkdir checkpoints\n!wget -c http://download.openmmlab.com/mmdetection/v2.0/cascade_rcnn/cascade_rcnn_r50_caffe_fpn_1x_coco/cascade_rcnn_r50_caffe_fpn_1x_coco_bbox_mAP-0.404_20200504_174853-b857be87.pth -O checkpoints/cascade_rcnn_r50_caffe_fpn_1x_coco_bbox_mAP-0.404_20200504_174853-b857be87.pth","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from mmcv import Config\nfrom mmdet.apis import set_random_seed\nfrom mmdet.datasets import build_dataset\nfrom mmdet.models import build_detector\nfrom mmdet.apis import train_detector, init_detector, inference_detector\n\nfrom IPython.display import clear_output","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Configuration Setting\ncfg = Config.fromfile('./configs/cascade_rcnn/cascade_rcnn_r50_fpn_1x_coco.py')\nDATASET_TYPE = 'CocoDataset'\nPREFIX = '../../input/artificial-vinbigdatacoco-format/'\ncfg.dataset_type = DATASET_TYPE\ncfg.classes = (\"Aortic_enlargement\", \"Atelectasis\", \n               \"Calcification\", \"Cardiomegaly\", \n               \"Consolidation\", \"ILD\", \"Infiltration\", \n               \"Lung_Opacity\", \"Nodule/Mass\", \"Other_lesion\", \n               \"Pleural_effusion\", \"Pleural_thickening\", \n               \"Pneumothorax\", \"Pulmonary_fibrosis\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cfg.data.train.img_prefix = PREFIX\ncfg.data.train.classes = cfg.classes\ncfg.data.train.ann_file = PREFIX + 'train_annotations.json'\ncfg.data.train.type = DATASET_TYPE\n\n\ncfg.data.val.img_prefix = PREFIX\ncfg.data.val.classes = cfg.classes\ncfg.data.val.ann_file = PREFIX + 'val_annotations.json'\ncfg.data.val.type = DATASET_TYPE\n\n\n\ncfg.data.test.img_prefix = PREFIX\ncfg.data.test.classes = cfg.classes\ncfg.data.test.ann_file = PREFIX + 'val_annotations.json'\ncfg.data.test.type = DATASET_TYPE","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in cfg.model.roi_head.bbox_head:\n    i.num_classes = 14\n    \n# for i in cfg.model.roi_head.mask_head:\n#     i.num_classes = 14","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cfg.optimizer.lr = 0.02 / 8\ncfg.lr_config.warmup = None\ncfg.log_config.interval = 100\n\n# Change the evaluation metric since we use customized dataset.\ncfg.evaluation.metric = 'bbox'\n# We can set the evaluation interval to reduce the evaluation times\ncfg.evaluation.interval = 5\n# We can set the checkpoint saving interval to reduce the storage cost\ncfg.checkpoint_config.interval = 5\n\n# Set seed thus the results are more reproducible\ncfg.seed = 0\nset_random_seed(0, deterministic=False)\ncfg.gpu_ids = range(1)\n\n# we can use here mask_rcnn.\ncfg.load_from = './checkpoints/cascade_rcnn_r50_caffe_fpn_1x_coco_bbox_mAP-0.404_20200504_174853-b857be87.pth'\ncfg.work_dir = \"../vinbig_output\"\n\ncfg.runner.max_epochs = 10\ncfg.total_epochs = 10","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clear_output()\nmodel = build_detector(cfg.model)\ndatasets = [build_dataset(cfg.data.train)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_detector(model, datasets[0], cfg, distributed=False, validate=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nos.chdir('../')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!python ./mmdetection/tools/analysis_tools/analyze_logs.py plot_curve ./vinbig_output/None.log.json --keys s2.loss_cls --legend s2.loss_cls --out \"loss_cls.jpg\"\n!rm -rf \"./mmdetection\"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}