{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# !pip install iterative-stratification\n!pip install mmcv-full","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Install mmcls\n!git clone https://github.com/open-mmlab/mmclassification.git\n%cd mmclassification\n!pip install -e .","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check Pytorch installation\nimport torch, torchvision\nprint(torch.__version__, torch.cuda.is_available())\n\n# Check MMClassification installation\nimport mmcls\nprint(mmcls.__version__)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!mkdir checkpoints\n# !wget https://download.openmmlab.com/mmclassification/v0/resnet/resnet50_batch256_imagenet_20200708-cfb998bf.pth -P checkpoints\n!wget https://download.openmmlab.com/mmclassification/v0/resnext/resnext50_32x4d_batch256_imagenet_20200708-c07adbb7.pth -P checkpoints","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from mmcls.apis import inference_model, init_model, show_result_pyplot\n# Specify the path to config file and checkpoint file\nconfig_file = 'configs/resnext/resnext101_32x4d_b32x8_imagenet.py'\ncheckpoint_file = 'checkpoints/resnet50_batch256_imagenet_20200708-cfb998bf.pth'\n# checkpoint_file = 'checkpoints/resnext50_32x4d_batch256_imagenet_20200708-c07adbb7.pth'\n# Specify the device. You may also use cpu by `device='cpu'`.\ndevice = 'cuda:0'\n# Build the model from a config file and a checkpoint file\nmodel = init_model(config_file, checkpoint_file, device=device)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Test a single image\nimg = 'demo/demo.JPEG'\nresult = inference_model(model, img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Show the results\nshow_result_pyplot(model, img, result)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv(\"../../input/vinbigdata-chest-xray-abnormalities-detection/train.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.image_id.nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_new = df.groupby('image_id')['class_id'].apply(list).reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_new.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_new['class_id'] = df_new['class_id'].apply(lambda x: list(set(x)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_new.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['label'] = 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.loc[df.class_name!='No finding', ['label']] = 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_with_labels = df.groupby('image_id')['label'].sum().reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_with_labels.loc[df_with_labels.label>0,['label']] = 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_with_labels.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(df_with_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nskf = StratifiedKFold(n_splits=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_ids, labels = np.array(df_with_labels.image_id.tolist()), np.array(df_with_labels.label.tolist())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for train_index, test_index in skf.split(image_ids, labels):\n    X_train, X_test = image_ids[train_index], image_ids[test_index]\n    y_train, y_test = labels[train_index], labels[test_index]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.DataFrame({'id': X_train, 'y': y_train})\nval_df = pd.DataFrame({'id': X_test, 'y': y_test})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df = pd.read_csv(\"../../input/vinbigdata-1024-image-dataset/vinbigdata/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df['image_id']  = test_df['image_id']  + \".png\"\ntest_df['label'] = 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df[['image_id', 'label']].to_csv('./test.txt', sep=' ', header=False, index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['id'] = train_df['id'] + \".png\"\nval_df['id'] = val_df['id'] + \".png\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.to_csv(\"./train.txt\", sep=\" \", header=False, index=False)\nval_df.to_csv(\"./val.txt\", sep=\" \", header=False, index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import mmcv\nimport numpy as np\n\nfrom mmcls.datasets import DATASETS, BaseDataset\n\n\n# Regist model so that we can access the class through str in configs\n@DATASETS.register_module()\nclass VinBigDataset(BaseDataset):\n\n    def load_annotations(self):\n        assert isinstance(self.ann_file, str)\n\n        data_infos = []\n        with open(self.ann_file) as f:\n            # The ann_file is the annotation files we generate above.\n            samples = [x.strip().split(' ') for x in f.readlines()]\n            for filename, gt_label in samples:\n                info = {'img_prefix': self.data_prefix}\n                info['img_info'] = {'filename': filename}\n                info['gt_label'] = np.array(gt_label, dtype=np.int64)\n                data_infos.append(info)\n            return data_infos","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\n# Load the existing config file\nfrom mmcv import Config\n# cfg = Config.fromfile('configs/resnet/resnet50_b32x8_imagenet.py')\ncfg = Config.fromfile('configs/resnext/resnext101_32x4d_b32x8_imagenet.py')\n\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os.path as osp\nclasses = ['normal', 'diseases']\nwith open(osp.join('./', 'classes.txt'), 'w') as f:\n    f.writelines('\\n'.join(classes))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Specify the new dataset class\ncfg.dataset_type = 'VinBigDataset'\ncfg.data.train.type = cfg.dataset_type\ncfg.data.val.type = cfg.dataset_type\ncfg.data.test.type = cfg.dataset_type\n\n# Specify the training annotations\ncfg.data.train.ann_file = './train.txt'\n\n# The followings are the same as above\ncfg.data.samples_per_gpu = 32\ncfg.data.workers_per_gpu=2\n\ncfg.img_norm_cfg = dict(\n    mean=[124.508, 116.050, 106.438], std=[58.577, 57.310, 57.437], to_rgb=True)\n\ncfg.data.train.data_prefix = '../../input/vinbigdata-1024-image-dataset/vinbigdata/train'\ncfg.data.train.classes = './classes.txt'\n\ncfg.data.val.data_prefix = '../../input/vinbigdata-1024-image-dataset/vinbigdata/train'\ncfg.data.val.ann_file = './val.txt'\ncfg.data.val.classes = './classes.txt'\n\ncfg.data.test.data_prefix = '../../input/vinbigdata-1024-image-dataset/vinbigdata/train'\ncfg.data.test.ann_file = './val.txt'\ncfg.data.test.classes = './classes.txt'\n# Modify the metric method\ncfg.evaluation['metric_options']={'topk': (1)}\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"! cat \"classes.txt\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# MODOL CONFIG\n# Modify num classes of the model in classification head\ncfg.model.head.num_classes = 2\ncfg.model.head.topk = (1)\n\n# SCHEDULE CONFIG\n# Optimizer\ncfg.optimizer = dict(type='SGD', lr=0.01, momentum=0.9, weight_decay=0.0001)\ncfg.optimizer_config = dict(grad_clip=None)\n# Learning policy\ncfg.lr_config = dict(policy='step', step=[1])\ncfg.runner = dict(type='EpochBasedRunner', max_epochs=2)\n\n# RUNTIME CONFIG\n# Load the pretrained weights\n# cfg.load_from = 'checkpoints/resnet50_batch256_imagenet_20200708-cfb998bf.pth'\ncfg.load_from = 'checkpoints/resnext50_32x4d_batch256_imagenet_20200708-c07adbb7.pth'\n# Set up working dir to save files and logs.\ncfg.work_dir = './vin_work_dirs'\nfrom mmcls.apis import set_random_seed\n# Set seed thus the results are more reproducible\ncfg.seed = 0\nset_random_seed(0, deterministic=False)\ncfg.gpu_ids = range(1)\n\n# Let's have a look at the final config used for training\nprint(f'Config:\\n{cfg.pretty_text}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cfg.runner.max_epochs = 20","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls ./vin_work_dirs","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import time\n\nfrom mmcls.datasets import build_dataset\nfrom mmcls.models import build_classifier\nfrom mmcls.apis import train_model\n\n# Create work_dir\nmmcv.mkdir_or_exist(osp.abspath(cfg.work_dir))\n# Build the classifier\nmodel = build_classifier(cfg.model)\n# Build the dataset\ndatasets = [build_dataset(cfg.data.train)]\n# Add an attribute for visualization convenience\nmodel.CLASSES = datasets[0].CLASSES\n# Begin finetuning\ntrain_model(\n    model,\n    datasets,\n    cfg,\n    distributed=False,\n    validate=True,\n    timestamp=time.strftime('%Y%m%d_%H%M%S', time.localtime()),\n    meta=dict())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img = mmcv.imread('../../input/vinbigdata-1024-image-dataset/vinbigdata/test/002a34c58c5b758217ed1f584ccbcfe9.png')\n\nmodel.cfg = cfg\nresult = inference_model(model, img)\nplt.figure(figsize=(8, 6))\nshow_result_pyplot(model, img, result)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# \"img_norm_cfg\":\"dict(mean=[124.508, 116.050, 106.438], std=[58.577, 57.310, 57.437], to_rgb=True)\"\n# \"evaluation.metric_options\": '''dict('topk' : '(1)')''',\n# \"optimizer\" :  '''dict('type'='SGD', 'lr'='0.01', 'momentum'='0.9', 'weight_decay'='0.0001')''',\n# \"lr_config\" : '''dict('policy'='step', 'step'='[1]')''',\n# \"runner\" : '''dict('type'='EpochBasedRunner', 'max_epochs'='2')''',\n\n# \"data.train.type\":'VinBigDataset',\n#     \"data.val.type\" : 'VinBigDataset',\n#         \"data.test.type\": 'VinBigDataset',   \n_cfg_options = {\n\"classes\" : './classes.txt',\n\"data.train.data_prefix\" : '../../input/vinbigdata-1024-image-dataset/vinbigdata/train/',\n\"data.train.classes\" : './classes.txt',\n\"data.train.ann_file\" : './train.txt',\n             \n\"data.val.data_prefix\" : '../../input/vinbigdata-1024-image-dataset/vinbigdata/test/',\n\"data.val.classes\" : './classes.txt',\n\"data.val.ann_file\" : './val.txt',\n\n\"data.test.data_prefix\" :  '../../input/vinbigdata-1024-image-dataset/vinbigdata/test/',\n\"data.test.classes\" :  './classes.txt',\n\"data.test.ann_file\" :  './test.txt',\n'evaluation.metric_options.topk' : '1',\n\"data.samples_per_gpu\":'32',\n\"data.workers_per_gpu\" :'2',\n\"model.head.num_classes\":'2',\n\"model.head.topk\":\"1\",\n\"optimizer.type\" : 'SGD',\n\"optimizer.lr\" : '0.01',\n\"optimizer.momentum\" : '0.9',\n\"optimizer.weight_decay\" : '0.0001',\n\"optimizer_config.grad_clip\" : \"None\",\n\"lr_config.policy\": 'step',\n\"lr_config.step\": '1',\n\"runner.type\": \"EpochBasedRunner\",\n\"runner.max_epochs\": \"2\",\n    \n\"load_from\" : 'checkpoints/resnext50_32x4d_batch256_imagenet_20200708-c07adbb7.pth',\n\"work_dir\" : './vin_work_dirs_val',\n\n\"runner.max_epochs\" : \"20\",\n   \n\"img_norm_cfg.to_rgb\" : 'True'\n}\n# \"load_from\" : 'checkpoints/resnet50_batch256_imagenet_20200708-cfb998bf.pth',\n# \"gpu_ids\" : \"range(0, 1)\",\n# \"img_norm_cfg.mean\" : \"[124.508, 116.050, 106.438]\",\n# \"img_norm_cfg.std\" : \"[58.577, 57.310, 57.437]\",  \ncfg_op = \"\"\nfor k, v in _cfg_options.items():\n    cfg_op+=f\"{k}='{v}' \"\nprint(cfg_op)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":""},{"metadata":{"trusted":true},"cell_type":"code","source":"!python tools/test.py configs/resnext/resnext101_32x4d_b32x8_imagenet.py ./vin_work_dirs/latest.pth --out=results_resnext_20_epoch.json --options classes='./classes.txt' data.train.data_prefix='../../input/vinbigdata-1024-image-dataset/vinbigdata/train/' data.train.classes='./classes.txt' data.train.ann_file='./train.txt' data.val.data_prefix='../../input/vinbigdata-1024-image-dataset/vinbigdata/test/' data.val.classes='./classes.txt' data.val.ann_file='./val.txt' data.test.data_prefix='../../input/vinbigdata-1024-image-dataset/vinbigdata/test/' data.test.classes='./classes.txt' data.test.ann_file='./test.txt' evaluation.metric_options.topk='1' data.samples_per_gpu='32' data.workers_per_gpu='2' model.head.num_classes='2' model.head.topk='1' optimizer.type='SGD' optimizer.lr='0.01' optimizer.momentum='0.9' optimizer.weight_decay='0.0001' optimizer_config.grad_clip='None' lr_config.policy='step' lr_config.step='1' runner.type='EpochBasedRunner' runner.max_epochs='20' load_from='checkpoints/resnext50_32x4d_batch256_imagenet_20200708-c07adbb7.pth' work_dir='./vin_work_dirs_val' img_norm_cfg.to_rgb='True'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# test_df","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}