{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"package_path = \"../input/efficientnet-pytorch/EfficientNet-PyTorch/EfficientNet-PyTorch-master/\"\nimport sys \nsys.path.append(package_path)\n#!pip install open_clip_torch","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:58:39.115073Z","iopub.execute_input":"2022-10-28T08:58:39.115459Z","iopub.status.idle":"2022-10-28T08:58:39.139563Z","shell.execute_reply.started":"2022-10-28T08:58:39.11533Z","shell.execute_reply":"2022-10-28T08:58:39.138503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install timm","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:58:39.141664Z","iopub.execute_input":"2022-10-28T08:58:39.14204Z","iopub.status.idle":"2022-10-28T08:58:52.352293Z","shell.execute_reply.started":"2022-10-28T08:58:39.141994Z","shell.execute_reply":"2022-10-28T08:58:52.35112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torchvision\nimport torch.nn.functional as F\nimport torch.nn as nn\nfrom torch.utils.data import DataLoader, Dataset, WeightedRandomSampler\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nfrom sklearn.metrics import accuracy_score, roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold, KFold\n#from efficientnet_pytorch import model as enet\nimport efficientnet_pytorch\n#from efficientnet_pytorch import EfficientNet\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nimport timm\nimport gc\nimport os\nimport cv2\nimport tifffile\nimport time\nimport torch.optim as optim\nimport torchvision.transforms as transforms\nfrom torchmetrics.functional import auc\nimport torchvision.models as models\nfrom tqdm import tqdm\nfrom collections import defaultdict\nimport pytorch_lightning as pl\nfrom sklearn import metrics\nimport torch.nn as nn\nimport datetime\nfrom torch.optim import lr_scheduler\nimport warnings\nimport random\nimport copy\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport albumentations as A\nfrom albumentations import (Normalize, VerticalFlip, HorizontalFlip, Compose,\n                            RandomBrightnessContrast, HueSaturationValue,\n                            RandomResizedCrop, ShiftScaleRotate)\nfrom albumentations.pytorch.transforms import ToTensorV2\n%matplotlib inline\n\nfrom colorama import Fore, Back, Style\nb_ = Fore.BLUE\nsr_ = Style.RESET_ALL","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:58:52.354524Z","iopub.execute_input":"2022-10-28T08:58:52.355163Z","iopub.status.idle":"2022-10-28T08:58:58.566399Z","shell.execute_reply.started":"2022-10-28T08:58:52.355118Z","shell.execute_reply":"2022-10-28T08:58:58.565393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CONFIG:\n    seed = 42\n    model_name = 'tf_efficientnetv2_s_in21k' \n    train_batch_size = 4\n    img_size = 456\n    epochs = 5\n    learning_rate = 1e-3\n    lr_factor = 0.4\n    min_lr = 1e-4\n    T_max = 10\n    scheduler = 'CosineAnnealingLR'\n    n_accumulate = 1\n    n_fold = 5\n    target_size = 1\n    device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:58:58.568872Z","iopub.execute_input":"2022-10-28T08:58:58.56951Z","iopub.status.idle":"2022-10-28T08:58:58.64014Z","shell.execute_reply.started":"2022-10-28T08:58:58.569472Z","shell.execute_reply":"2022-10-28T08:58:58.638386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed = 42):\n    '''Sets the seed of the entire notebook so results are the same every time we run.\n    This is for REPRODUCIBILITY.'''\n    np.random.seed(seed)\n    random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    # When running on the CuDNN backend, two further options must be set\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    # Set a fixed value for the hash seed\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    \nset_seed(CONFIG.seed)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:58:58.641484Z","iopub.execute_input":"2022-10-28T08:58:58.642012Z","iopub.status.idle":"2022-10-28T08:58:58.657808Z","shell.execute_reply.started":"2022-10-28T08:58:58.641982Z","shell.execute_reply":"2022-10-28T08:58:58.656889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DIR ='../input/jpg-images-strip-ai/train'\nTEST_DIR = '../input/jpg-images-strip-ai/test'","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:58:58.658968Z","iopub.execute_input":"2022-10-28T08:58:58.66058Z","iopub.status.idle":"2022-10-28T08:58:58.666749Z","shell.execute_reply.started":"2022-10-28T08:58:58.660544Z","shell.execute_reply":"2022-10-28T08:58:58.665877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Creating the data class**","metadata":{}},{"cell_type":"code","source":"class MayoClinicDataset(Dataset):\n    def __init__(self,df, train_val = True, transforms = None):\n        self.df = df\n        self.file_names = df.file_path.values\n        self.train_val = train_val\n        self.train = train \n        self.transforms = transforms\n        if train_val:             \n            self.label = df.label.values\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        path = self.file_names[index]\n        image = cv2.imread(path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        if self.transforms:\n            image = self.transforms(image = image)['image']\n        lab = None\n        if self.train_val:\n            if self.label[index] == 'CE':\n                lab = 0\n            else:\n                lab = 1\n            return image, lab\n        else:\n            return image","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:58:58.668226Z","iopub.execute_input":"2022-10-28T08:58:58.668709Z","iopub.status.idle":"2022-10-28T08:58:58.678901Z","shell.execute_reply.started":"2022-10-28T08:58:58.668675Z","shell.execute_reply":"2022-10-28T08:58:58.677857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Image Transforms**","metadata":{}},{"cell_type":"code","source":"transforms = A.Compose([\n        A.Resize(CONFIG.img_size, CONFIG.img_size),\n        A.ShiftScaleRotate(shift_limit=0.1, \n                           scale_limit=0.15, \n                           rotate_limit=60, \n                           p=0.5),\n        A.HueSaturationValue(\n                hue_shift_limit=0.2, \n                sat_shift_limit=0.2, \n                val_shift_limit=0.2, \n                p=0.5\n            ),\n        A.RandomBrightnessContrast(\n                brightness_limit=(-0.1,0.1), \n                contrast_limit=(-0.1, 0.1), \n                p=0.5\n            ),\n        A.Normalize(\n                mean=[0.485, 0.456, 0.406], \n                std=[0.229, 0.224, 0.225], \n                max_pixel_value=255.0, \n                p=1.0\n            ),\n        ToTensorV2()], p=1.)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:58:58.680226Z","iopub.execute_input":"2022-10-28T08:58:58.681133Z","iopub.status.idle":"2022-10-28T08:58:58.69009Z","shell.execute_reply.started":"2022-10-28T08:58:58.681097Z","shell.execute_reply":"2022-10-28T08:58:58.689172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model Class**\nModel used- EfficientNet","metadata":{}},{"cell_type":"code","source":"class EffModel(nn.Module):\n    def __init__(self, backbone):\n        super(EffModel, self).__init__()\n        self.net = timm.create_model(backbone, pretrained = True)\n        n_features = self.net.classifier.out_features\n        self.bn1 = nn.BatchNorm1d(n_features)\n        self.dropout = nn.Dropout(0.2)\n        self.linear = nn.Linear(n_features, 2)\n        self.softmax = nn.Sigmoid()\n        \n    def forward(self, x):\n        x = self.net(x)\n        x = self.bn1(x)\n        x = self.dropout(x)\n        x = self.linear(x)\n        return self.softmax(x)\nmodel = EffModel('efficientnet_b5')","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:58:58.693413Z","iopub.execute_input":"2022-10-28T08:58:58.693854Z","iopub.status.idle":"2022-10-28T08:58:59.233683Z","shell.execute_reply.started":"2022-10-28T08:58:58.693817Z","shell.execute_reply":"2022-10-28T08:58:59.232677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\ntrain_df = train_df[train_df['label'] != -1]\ntrain_df['image_id'] = train_df.image_id.apply(lambda x: f\"{x}.jpg\")\ntrain_df['file_path'] = train_df.image_id.apply(lambda x: os.path.join(TRAIN_DIR, x))\n\ntrain, val = train_test_split(train_df, test_size=0.2, random_state=42, stratify = train_df.label)\n\ntrain_data = MayoClinicDataset(train, train_val = True, transforms = transforms)\nvalid_data = MayoClinicDataset(val, transforms)\n\n\noptimizer = torch.optim.Adam(model.parameters(), lr=CONFIG.learning_rate)\n\n#criterion = nn.BCEWithLogitsLoss()\nscheduler = lr_scheduler.ReduceLROnPlateau(optimizer, factor = 0.1,patience = 10, mode = 'min' , threshold = 0.0000001)\ncriterion = nn.CrossEntropyLoss()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:58:59.235409Z","iopub.execute_input":"2022-10-28T08:58:59.23576Z","iopub.status.idle":"2022-10-28T08:58:59.274625Z","shell.execute_reply.started":"2022-10-28T08:58:59.235723Z","shell.execute_reply":"2022-10-28T08:58:59.273786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Using WeightedRandomSampler To Handle Unbalanced Data**","metadata":{}},{"cell_type":"code","source":"def loader(data):\n    class_weights = [1/2.9, 1]\n    sample_weights = [0]*len(data)\n    \n    for idx, (_, label) in enumerate(data):\n        class_weight = class_weights[label]\n        sample_weights[idx] = class_weight\n    \n    sampler = WeightedRandomSampler(sample_weights, num_samples = len(sample_weights), replacement = True)\n    loader = DataLoader(train_data, batch_size = CONFIG.train_batch_size, shuffle = False, num_workers=1, pin_memory=False, sampler = sampler)\n    \n    return loader","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:58:59.275862Z","iopub.execute_input":"2022-10-28T08:58:59.276298Z","iopub.status.idle":"2022-10-28T08:58:59.282816Z","shell.execute_reply.started":"2022-10-28T08:58:59.276262Z","shell.execute_reply":"2022-10-28T08:58:59.281835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader = loader(train_data)\nval_loader = loader(valid_data)\ndataloaders_dict = {\"train\": train_loader, \"val\": val_loader}","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:58:59.284126Z","iopub.execute_input":"2022-10-28T08:58:59.284734Z","iopub.status.idle":"2022-10-28T08:59:08.804875Z","shell.execute_reply.started":"2022-10-28T08:58:59.284691Z","shell.execute_reply":"2022-10-28T08:59:08.803713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Accuracy**","metadata":{}},{"cell_type":"code","source":"def correct_to_list(output, labels):\n    #correct = 0\n    correct = []\n    for i, sample in enumerate(output):\n        result = torch.argmax(sample)\n        if result == labels[i]:\n            #correct+=1\n            correct.append(1)\n        else:\n            correct.append(0)\n    return correct","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:59:08.806517Z","iopub.execute_input":"2022-10-28T08:59:08.808591Z","iopub.status.idle":"2022-10-28T08:59:08.814391Z","shell.execute_reply.started":"2022-10-28T08:59:08.808561Z","shell.execute_reply":"2022-10-28T08:59:08.813272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Training Pipeline**","metadata":{}},{"cell_type":"code","source":"def train_one_epoch(model, dataloader, criterion, optimizer, epoch, scheduler):\n    model.train()\n    gc.collect()\n    \n    \n    dataset_size = 0.0\n    running_loss = 0.0\n    correct = []\n    label = []\n    \n    bar = tqdm(dataloader, total = len(dataloader))\n    for step, (images, labels) in enumerate(bar):\n        images = images.cuda().float()\n        labels = labels.cuda()\n        \n        batch = images.size(0)\n        \n        output = model(images,)\n        loss = criterion(output, labels)\n            \n        loss.backward()\n        \n        if (step + 1) % CONFIG.n_accumulate == 0:\n            optimizer.step()\n\n            # zero the parameter gradients\n            optimizer.zero_grad()  \n            \n        \n        \n        running_loss += loss.item()*batch\n        correct += correct_to_list(output, labels)\n        label_list = labels.tolist()\n        label += label_list\n        dataset_size += batch\n        \n        epoch_loss = running_loss/dataset_size\n        \n        scheduler.step(epoch_loss)\n        \n        bar.set_postfix(Epoch = epoch, Train_loss = epoch_loss, LR = optimizer.param_groups[0]['lr'])\n    \n    auc = metrics.roc_auc_score(label, correct)\n    gc.collect()\n    return epoch_loss, auc","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:59:08.816755Z","iopub.execute_input":"2022-10-28T08:59:08.81778Z","iopub.status.idle":"2022-10-28T08:59:08.833369Z","shell.execute_reply.started":"2022-10-28T08:59:08.817742Z","shell.execute_reply":"2022-10-28T08:59:08.832265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Validation Pipeline**","metadata":{}},{"cell_type":"code","source":"def valid_one_epoch(model, dataloader, optimizer, criterion, epoch):\n    model.eval()\n    gc.collect()\n    \n    dataset_size = 0.0\n    running_loss = 0.0\n    \n    bar = tqdm(dataloader, total = len(dataloader))\n    for step, (images, labels) in enumerate(bar):\n        images = images.cuda().float()\n        labels = labels.cuda()\n        \n        batch = images.size(0)\n        \n        output = model(images)\n        loss = criterion(output, labels)\n        \n        running_loss += loss.item()*batch\n        dataset_size += batch\n        \n        \n        epoch_loss = running_loss/dataset_size\n        \n        bar.set_postfix(Epoch=epoch, Valid_Loss=epoch_loss,\n                        LR=optimizer.param_groups[0]['lr'])\n    \n    gc.collect()\n    return epoch_loss","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:59:08.836056Z","iopub.execute_input":"2022-10-28T08:59:08.836347Z","iopub.status.idle":"2022-10-28T08:59:08.847966Z","shell.execute_reply.started":"2022-10-28T08:59:08.836314Z","shell.execute_reply":"2022-10-28T08:59:08.847088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_training(model, optimizer, scheduler, num_epochs):\n    model.to('cuda')\n    start = time.time()\n    best_model_wts = copy.deepcopy(model.state_dict())\n    best_auc = 0\n    best_file = f'best_model.pth'\n    history = defaultdict(list)\n    \n    for epoch in range(num_epochs):\n        train_epoch_loss, train_auc = train_one_epoch(model, train_loader, criterion, optimizer, epoch, scheduler)\n        val_epoch_loss = valid_one_epoch(model, val_loader, optimizer, criterion, epoch)\n        history['Train Loss'].append(train_epoch_loss)\n                \n        history['Valid Loss'].append(val_epoch_loss)\n        \n        print(f'Epoch {epoch + 1}/{num_epochs} | Train Loss: {val_epoch_loss} | Train AUC: {train_auc}')\n        \n        if train_auc > best_auc:\n            print('Saving model ...')\n            torch.save(model.state_dict(), best_file)\n            best_auc = train_auc\n          \n        print()\n    end = time.time()\n    time_elapsed = end - start\n    print('Training complete in {:.0f}h {:.0f}m {:.0f}s'.format(\n        time_elapsed // 3600, (time_elapsed % 3600) // 60, (time_elapsed % 3600) % 60))\n    print(\"Best Loss: {:.4f}\".format(val_epoch_loss))\n    \n    torch.save(model.state_dict(), os.path.join(f'final_model.pth'))\n    \n    return model, history","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:59:08.849423Z","iopub.execute_input":"2022-10-28T08:59:08.849945Z","iopub.status.idle":"2022-10-28T08:59:08.861284Z","shell.execute_reply.started":"2022-10-28T08:59:08.849912Z","shell.execute_reply":"2022-10-28T08:59:08.860237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel, history = run_training(model, optimizer, scheduler, 5)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T08:59:08.862946Z","iopub.execute_input":"2022-10-28T08:59:08.863352Z","iopub.status.idle":"2022-10-28T09:03:13.374688Z","shell.execute_reply.started":"2022-10-28T08:59:08.86332Z","shell.execute_reply":"2022-10-28T09:03:13.37363Z"},"trusted":true},"execution_count":null,"outputs":[]}]}