{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nfrom torch.nn import BCEWithLogitsLoss\nimport torchvision\nfrom torchvision import transforms\nfrom torch.utils.data import random_split,DataLoader, Dataset, WeightedRandomSampler\nfrom sklearn import model_selection\nimport os\nimport PIL\nfrom PIL import Image\nfrom tqdm import tqdm\n\n\ntorch.manual_seed(17)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-27T08:07:23.841517Z","iopub.execute_input":"2023-04-27T08:07:23.842122Z","iopub.status.idle":"2023-04-27T08:07:23.853327Z","shell.execute_reply.started":"2023-04-27T08:07:23.842051Z","shell.execute_reply":"2023-04-27T08:07:23.852238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' There are three methods which allow overcoming imbalanced data, they are as follows :\nFIRST METHOD : Augmentations\nThis is a technique used to increase the size of training dataset, by applying tranfsormations on images \nand so, creating new images. It allows improving the model perfomance. In the case of imbalanced data, we can use this technique to generate \nnew images that belong to the minority class.\n\nSECOND METHOD : Over-sampling the dataset by using WeightedRandomSampler (In this example, \nthe positive class is the minority class). This technique consists in calculating a weight to each \nclass, and assign this weight to each element in the dataset.\n\nTHIRD METHOD : loss weights : \nThis method consists in calculating a manual re-scaling weight for each class and pass it to \n“weight” parameter in the loss function.\n\n'''","metadata":{"execution":{"iopub.status.busy":"2023-04-27T08:07:23.855757Z","iopub.execute_input":"2023-04-27T08:07:23.856954Z","iopub.status.idle":"2023-04-27T08:07:23.864457Z","shell.execute_reply.started":"2023-04-27T08:07:23.856912Z","shell.execute_reply":"2023-04-27T08:07:23.863338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' In order to apply these methods, we use as example the dataset of RSNA competition \n(screening mammography Breast cancer detection). Let us start by importing our dataser and defining \nour dataset class.\n\n'''","metadata":{"execution":{"iopub.status.busy":"2023-04-27T08:07:23.866635Z","iopub.execute_input":"2023-04-27T08:07:23.866985Z","iopub.status.idle":"2023-04-27T08:07:23.8782Z","shell.execute_reply.started":"2023-04-27T08:07:23.86695Z","shell.execute_reply":"2023-04-27T08:07:23.877059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv_path = '/kaggle/input/rsna-breast-cancer-detection/train.csv'\ntest_csv_path = '/kaggle/input/rsna-breast-cancer-detection/test.csv'\n\ndfx = pd.read_csv(train_csv_path)\ndf_test = pd.read_csv(test_csv_path)","metadata":{"execution":{"iopub.status.busy":"2023-04-27T08:07:23.879914Z","iopub.execute_input":"2023-04-27T08:07:23.880765Z","iopub.status.idle":"2023-04-27T08:07:23.939306Z","shell.execute_reply.started":"2023-04-27T08:07:23.880724Z","shell.execute_reply":"2023-04-27T08:07:23.938149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass Dataset:\n    def __init__(self, df, transform):\n        self.df = df.copy()\n        self.transform = transform\n     \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        \n        patient_id = self.df.loc[idx, 'patient_id']\n        image_id = self.df.loc[idx, 'image_id']\n        # Target\n        target = self.df.loc[idx, 'cancer'] \n        \n        \n        # File path\n        png_path = '/kaggle/input/png-cutted-images/output/png_cutted_images'\n        \n        # Image path\n        image_png_path =  os.path.join(png_path, patient_id.astype(str), image_id.astype(str)+'.png')\n\n        # convert image to RGB\n        image = Image.open(image_png_path).convert('RGB')\n        \n        # Check if image is RGB \n        #print('image shape = ', np.array(image).shape)\n        \n        # Resize image\n        image = image.resize((1024,912)) \n        \n        \n        # Apply transformers on images whose target equals to 1 (minority class)\n        if (target == 1) and self.transform:\n            image = self.transform(image).to(torch.float32)\n        else :\n            default_transform = transforms.Compose([transforms.ToTensor()])\n            \n            image = default_transform(image).to(torch.float32)\n            \n            \n                \n        \n        # Convert to tensors\n        #image = torch.tensor(image, dtype=torch.float32)\n        target = torch.tensor(target, dtype=torch.long)\n        \n        return image, target","metadata":{"execution":{"iopub.status.busy":"2023-04-27T08:07:23.945674Z","iopub.execute_input":"2023-04-27T08:07:23.945967Z","iopub.status.idle":"2023-04-27T08:07:23.955699Z","shell.execute_reply.started":"2023-04-27T08:07:23.945939Z","shell.execute_reply":"2023-04-27T08:07:23.954541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfx.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-27T08:07:23.957517Z","iopub.execute_input":"2023-04-27T08:07:23.958281Z","iopub.status.idle":"2023-04-27T08:07:23.982357Z","shell.execute_reply.started":"2023-04-27T08:07:23.958239Z","shell.execute_reply":"2023-04-27T08:07:23.980962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' FIRST METHOD : Augmentations '''\n\n# Empty Transforms\ntransform = transforms.Compose([\n    transforms.ToTensor()]\n)\n\n\n# Example of augmentations\naugmentator = transforms.Compose([\n    # input for augmentator is always PIL image\n    # transforms.ToPILImage(),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomVerticalFlip(),\n    transforms.RandomAffine(degrees=(0, 180), scale=(0.8, 1.2)),\n    transforms.ElasticTransform(),\n    transforms.ToTensor(), # return it as a tensor and transforms it to [0, 1]\n])\n\n'''\nPILToTensor transforms a PIL.Image to a PyTorch tensor of the same dtype while ToTensor normalizes \nthe tensor to [0, 1] and will return a FloatTensor.\nUsually normalized/standardized tensors are preferred while training a model as the convergence \noften benefits from it.\n\n'''\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-27T08:07:23.984412Z","iopub.execute_input":"2023-04-27T08:07:23.984883Z","iopub.status.idle":"2023-04-27T08:07:23.998468Z","shell.execute_reply.started":"2023-04-27T08:07:23.984846Z","shell.execute_reply":"2023-04-27T08:07:23.997439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' Apply augmentations '''\n# If you want to apply augmentations, you just need to define the transform parameter in the Dataset\n# class, as follows:\n\naugmented_dataset = Dataset(dfx, transform=augmentator)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-27T08:59:39.005884Z","iopub.execute_input":"2023-04-27T08:59:39.006849Z","iopub.status.idle":"2023-04-27T08:59:39.015587Z","shell.execute_reply.started":"2023-04-27T08:59:39.006793Z","shell.execute_reply":"2023-04-27T08:59:39.014287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split data using random split\n\ndataset = Dataset(dfx, transform=transform)\nval_pct = 0.1\nval_size = int(val_pct * len(dataset))\ntrain_size = len(dataset) - val_size\ndf_train, df_valid = random_split(dataset, [train_size, val_size])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' SECOND METHOD : WeightedRandomSampler '''\n\nprint(\"Class counting...\")\nlabels = dfx['cancer'].values\nclass_sample_count = np.array([len(np.where(labels == l)[0]) for l in np.unique(labels)])\n\nprint(class_sample_count)\n\nprint(\"Adding weights to each training sample...\")\n# class_sample_count[1] *= 5 # ??\nclass_weights = 1. / class_sample_count\nsample_weights = []\nfor _, label in tqdm(df_train):\n    sample_weights.append(class_weights[label])\n    \n    \n# the trouble with this aproach is that it now has to load all images one by one and label them\n# but it saves RAM memory in training process    \n    \nsample_weights = np.array(sample_weights)\nsample_weights = torch.from_numpy(sample_weights)\n\nweighted_random_sampler = WeightedRandomSampler(sample_weights, len(sample_weights),\n                                               replacement=True)\n\n# Now, we just need to define the sampler parameter in the DataLoader.\n","metadata":{"execution":{"iopub.status.busy":"2023-04-27T08:07:24.000142Z","iopub.execute_input":"2023-04-27T08:07:24.001224Z","iopub.status.idle":"2023-04-27T08:47:59.554464Z","shell.execute_reply.started":"2023-04-27T08:07:24.001184Z","shell.execute_reply":"2023-04-27T08:47:59.553229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the DataLoaders (and define the sampler parameter)\nbatch_size = 4 \ntrain_Dataloader = DataLoader(df_train, batch_size = batch_size, shuffle = False, \n                              num_workers = 2, drop_last=True, sampler = weighted_random_sampler)\n\nvalid_DalaLoader = DataLoader(df_valid, batch_size = batch_size, shuffle = False,\n                             num_workers = 2, drop_last=True)\n\nimages, targets = next(iter(train_Dataloader))\n\n#images, targets\n\n","metadata":{"execution":{"iopub.status.busy":"2023-04-27T09:15:15.214474Z","iopub.execute_input":"2023-04-27T09:15:15.214861Z","iopub.status.idle":"2023-04-27T09:15:24.418094Z","shell.execute_reply.started":"2023-04-27T09:15:15.214827Z","shell.execute_reply":"2023-04-27T09:15:24.416829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' THIRD METHOD : Weighted loss '''\n# Calculating weights for classes\n\n# Number of samples for each class\nsamples_classes = dfx['cancer'].value_counts()\nprint('samples_classes = ', samples_classes)\n\n\n# Total number of samples\ntotal_samples = dfx['cancer'].count()\nprint('total_samples = ', total_samples)\n\n# Class weight\nnegative_weight = 1 - (samples_classes[0] / total_samples)\npositive_weight = 1 - (samples_classes[1] / total_samples)\nprint('negative_weight = ', negative_weight)\nprint('positive_weight = ', positive_weight)\n\n# class weights for our 2-class classification\nclass_weights = [negative_weight, positive_weight]\nclass_weights = torch.tensor(class_weights).float()\nprint('class_weights = ', class_weights)\n\n# Loss function with class weights\ncriterion = nn.BCEWithLogitsLoss(weight = class_weights, reduction='mean')\n\n\n# Example to apply the weighted loss function\nBATCH_SIZE = 5\nn_classes = 2\n#output = torch.randn(BATCH_SIZE, n_classes)\noutput = torch.tensor([[-0.5013, -2.4151],\n        [ 1.1817,  1.0621],\n        [ 2.0916, -0.6872],\n        [-2.2757, -0.0477],\n        [-1.3668,  0.4466]], requires_grad=True)\n\n#target = torch.empty(BATCH_SIZE, n_classes).random_(2)\ntarget = torch.tensor([[ 0.8274, -0.7168],\n        [-1.9719,  0.2166],\n        [-0.6234, -2.6189],\n        [ 0.2475,  1.3175],\n        [ 0.0304, -0.4510]])\n\n\nloss = criterion(output, target)\n# loss = loss.mean() # if reduction is 'none'\nprint(loss)\n\n# Note that, in case of BCE loss we have just weight parameter. In case of BCEWithLogitsLoss, we have 2 \n# paremeters that we can use; pos_weight and weight parameters.\n\n# For BCEWithLogitsLoss pos_weight should be a torch.tensor of size=1:\n# For example, if a dataset contains 100 positive and 300 negative examples of a single class, \n# then pos_weight for the class should be equal to 300/100=3.\n\n\n''' References : \nhttps://naadispeaks.blog/2021/07/31/handling-imbalanced-classes-with-weighted-loss-in-pytorch/ \n\nhttps://discuss.pytorch.org/t/dealing-with-imbalanced-datasets-in-pytorch/22596/4\n\n'''\n\n'''' Note that :\nCE loss performs so poorly in the case of imbalance; see: \nhttps://stackoverflow.com/questions/71462326/pytorch-bcewithlogitsloss-calculating-pos-weight\n\n'''\n","metadata":{"execution":{"iopub.status.busy":"2023-04-27T09:17:28.781873Z","iopub.execute_input":"2023-04-27T09:17:28.782572Z","iopub.status.idle":"2023-04-27T09:17:28.811145Z","shell.execute_reply.started":"2023-04-27T09:17:28.782524Z","shell.execute_reply":"2023-04-27T09:17:28.810075Z"},"trusted":true},"execution_count":null,"outputs":[]}]}