{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install timm -q\n\nfrom IPython.display import display_html\ndef restartkernel() :\n    display_html(\"\",raw=True)\nrestartkernel()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-31T13:36:46.517085Z","iopub.execute_input":"2023-01-31T13:36:46.517747Z","iopub.status.idle":"2023-01-31T13:37:00.152189Z","shell.execute_reply.started":"2023-01-31T13:36:46.517711Z","shell.execute_reply":"2023-01-31T13:37:00.150707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\nimport torch\n\nimport torch.nn as nn\nfrom torchvision import transforms\n\nimport os\n\nimport pydicom as dicom\nimport cv2\n\nimport timm\nimport torch.optim as optim\nfrom sklearn import model_selection\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import StratifiedGroupKFold\n\nfrom tqdm.autonotebook import tqdm\n\nimport gc\ngc.collect()\ntorch.cuda.empty_cache()\n","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:00.155094Z","iopub.execute_input":"2023-01-31T13:37:00.155587Z","iopub.status.idle":"2023-01-31T13:37:00.330927Z","shell.execute_reply.started":"2023-01-31T13:37:00.155538Z","shell.execute_reply":"2023-01-31T13:37:00.329952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For a given patient_id explore all the images in the folder\npatient_id = '10006'\n\n#List all images for each image_id\n#list_dcm_images = os.listdir('/kaggle/input/rsna-breast-cancer-detection/train_images/'+patient_id)\npng_path = '/kaggle/input/rsna-png-images-same-format-as-original/output/rsna_pngs/train_images'\n\nlist_images = os.listdir(png_path + '/' + patient_id)\n\nfig = plt.figure(figsize=(10,10))\ni=0\nfor image_png in list_images:\n    \n    image_png_path = png_path + '/' + patient_id + '/' + image_png\n    image = mpimg.imread(image_png_path)\n    \n    image = cv2.resize(image, (512,512))\n    \n    ax = fig.add_subplot(2, 2, i+1)\n    i = i + 1\n    \n    ax.imshow(image)\n    ax.set_title('image_png')\n    ","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:00.333148Z","iopub.execute_input":"2023-01-31T13:37:00.333912Z","iopub.status.idle":"2023-01-31T13:37:01.040808Z","shell.execute_reply.started":"2023-01-31T13:37:00.333876Z","shell.execute_reply":"2023-01-31T13:37:01.03979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass Dataset:\n    def __init__(self, df, transform=None):\n        self.df = df.copy()\n        self.transform = transform\n     \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        \n        patient_id = self.df.loc[idx, 'patient_id']\n        image_id = self.df.loc[idx, 'image_id']\n        \n        # Get and preprocess images\n        #dcm_path = '/kaggle/input/rsna-breast-cancer-detection/train_images/'\n        #image = dicom.dcmread(image_path)\n        \n        png_path = '/kaggle/input/rsna-png-images-same-format-as-original/output/rsna_pngs/train_images'\n        \n        # Image path\n        #image_dcm_path =  os.path.join(png_path, patient_id.astype(str), image_id.astype(str)+'.dcm')\n        image_png_path =  os.path.join(png_path, patient_id.astype(str), image_id.astype(str)+'.png')\n\n        image = mpimg.imread(image_png_path)\n        \n        image = cv2.resize(image, (512,512))\n        \n        # Apply transformers on images\n        if self.transform:\n            image = self.transform(image)\n            \n        # Target\n        target = self.df.loc[idx, 'cancer']  \n        \n        # Convert to tensors\n        image = torch.tensor(image, dtype=torch.float)\n        target = torch.tensor(target, dtype=torch.long)\n        \n        return image, target","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:01.043668Z","iopub.execute_input":"2023-01-31T13:37:01.044027Z","iopub.status.idle":"2023-01-31T13:37:01.053173Z","shell.execute_reply.started":"2023-01-31T13:37:01.043988Z","shell.execute_reply":"2023-01-31T13:37:01.052009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    NUM_CLASSES = 1\n    MODEL_PATH = 'model.bin'\n    TRAIN_BATCH_SIZE = 6 #6\n    VALID_BATCH_SIZE = 5 #5\n    EPOCHS = 2\n    ","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:01.054692Z","iopub.execute_input":"2023-01-31T13:37:01.055579Z","iopub.status.idle":"2023-01-31T13:37:01.069678Z","shell.execute_reply.started":"2023-01-31T13:37:01.05554Z","shell.execute_reply":"2023-01-31T13:37:01.068744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' Model (we need to improve our model''' \nclass BreastCancerModel(nn.Module):\n    def __init__(self, Config):\n        super().__init__()\n        self.efficientnet = timm.create_model('efficientnet_b4', pretrained=True,\n                                             in_chans=1)\n        in_features = self.efficientnet.classifier.in_features\n        self.efficientnet.classifier = nn.Linear(in_features, Config.NUM_CLASSES)   \n        \n    def forward(self, image):\n        output = self.efficientnet(image)\n        \n        return output\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:01.071201Z","iopub.execute_input":"2023-01-31T13:37:01.071885Z","iopub.status.idle":"2023-01-31T13:37:01.082867Z","shell.execute_reply.started":"2023-01-31T13:37:01.071851Z","shell.execute_reply":"2023-01-31T13:37:01.081989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''Training function'''\ndef train(data_loader, model, optimizer, device, scheduler=None):\n    model.train()\n    tk0 = tqdm(data_loader, total=len(data_loader))\n    \n    for batch_idx, data in enumerate(tk0):\n        images = data[0]\n        targets = data[1]\n        \n        images = images.to(device, dtype=torch.float)\n        targets = targets.to(device, dtype=torch.long)\n        \n        outputs = model(images).squeeze()\n        \n        optimizer.zero_grad()\n    \n        loss = criterion(outputs, targets)\n        \n        loss.backward()\n        optimizer.step()\n        ","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:01.085293Z","iopub.execute_input":"2023-01-31T13:37:01.085979Z","iopub.status.idle":"2023-01-31T13:37:01.094403Z","shell.execute_reply.started":"2023-01-31T13:37:01.085938Z","shell.execute_reply":"2023-01-31T13:37:01.093443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''Evaluation function'''\n\ndef evaluation(data_loader, model, device):\n    model.eval()\n    \n    tar = []\n    predictions = []\n    for batch_idx, data in enumerate(data_loader):\n           \n            images = data[0]\n            targets = data[1]\n        \n            images = images.to(device, dtype=torch.float) # or long?\n            targets = targets.to(device, dtype=torch.long)\n       \n            with torch.no_grad():\n                outputs = model(images).squeeze()\n                loss = criterion(outputs, targets)\n                \n                #Cross entropy loss\n                #_, preds = torch.max(outputs, 1)\n                #predictions.append(preds.cpu().numpy())\n                \n                #loss bcewith logits\n                predictions.append(outputs.sigmoid().cpu().numpy())\n                \n                #target\n                tar.append(targets.cpu().numpy())\n            \n    predictions = np.concatenate(predictions) # this line convert the list \n    # of lists into a 1d array.\n    tar = np.concatenate(tar)\n\n    return predictions, tar     \n        \n    ","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:01.095816Z","iopub.execute_input":"2023-01-31T13:37:01.09617Z","iopub.status.idle":"2023-01-31T13:37:01.110882Z","shell.execute_reply.started":"2023-01-31T13:37:01.096135Z","shell.execute_reply":"2023-01-31T13:37:01.109918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''Loss'''\n#criterion = nn.CrossEntropyLoss()\n#criterion = nn.BCELoss()\n#criterion = nn.BCEWithLogitsLoss()\n\n#class_weight = torch.tensor([0.02, 0.98])\n\npos_weight = torch.tensor([46]).cuda()\n#pos_weight = torch.tensor([34]).cuda()\n#pos_weight = torch.tensor([20]).cuda()\ncriterion = torch.nn.BCEWithLogitsLoss(pos_weight=pos_weight)\n#criterion = nn.CrossEntropyLoss()\n","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:01.11233Z","iopub.execute_input":"2023-01-31T13:37:01.112673Z","iopub.status.idle":"2023-01-31T13:37:01.124193Z","shell.execute_reply.started":"2023-01-31T13:37:01.112641Z","shell.execute_reply":"2023-01-31T13:37:01.123197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' Probabilistic F1 score'''\ndef pfbeta(labels, predictions, beta):\n    y_true_count = 0\n    ctp = 0\n    cfp = 0\n\n    for idx in range(len(labels)):\n        prediction = min(max(predictions[idx], 0), 1)\n        if (labels[idx]):\n            y_true_count += 1\n            ctp += prediction\n        else:\n            cfp += prediction\n\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:01.128378Z","iopub.execute_input":"2023-01-31T13:37:01.128727Z","iopub.status.idle":"2023-01-31T13:37:01.136526Z","shell.execute_reply.started":"2023-01-31T13:37:01.128702Z","shell.execute_reply":"2023-01-31T13:37:01.135336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''Running'''\ndef run(df_train, df_valid):\n    #dfx = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv').dropna(axis=0).reset_index(drop=True)\n    \n    # Split data to training and validation data\n    #df_train, df_valid= model_selection.train_test_split(dfx, test_size=0.15, \n                                                   #random_state=42, \n                                                 #stratify=dfx['cancer'].values)\n    #df_train = df_train.reset_index(drop=True)\n    #df_valid = df_valid.reset_index(drop=True)\n    \n    # Transforms on images\n    # MAKE THEM HERE!!!!!\n    transform = transforms.Compose([\n    transforms.ToTensor()\n    ])\n    \n    # Instantiate Dataset with training data\n    train_dataset = Dataset(df_train, transform=transform)\n    \n    # Instantiate Dataloader with training dataset\n    train_data_loader = torch.utils.data.DataLoader(train_dataset, \n                                                    batch_size=Config.TRAIN_BATCH_SIZE, \n                                                    num_workers=1,\n                                                    shuffle=True,\n                                                    drop_last=True)\n    \n    # Instantiate Dataset with validation data\n    valid_data =  Dataset(df_valid, transform=transform)\n    \n    # Instantiate Dataloader with valiation dataset\n    valid_data_loader = torch.utils.data.DataLoader(valid_data, \n                                                    batch_size=Config.VALID_BATCH_SIZE,\n                                                    num_workers=1, \n                                                    shuffle=False,\n                                                    drop_last=True)\n    \n    # Set device as `cuda` (GPU)\n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\n    # Load pretrained model (we need to improve our model!!!)\n    #model = timm.create_model('efficientnet_b4', pretrained=True, in_chans=1)\n    model = BreastCancerModel(Config=Config)\n    \n    # Move the model to the GPU\n    model.to(device)\n    \n    \n    # The optimizer\n    params = model.parameters()\n    optimizer = optim.AdamW(params=params, lr=1e-4) \n    \n    \n    dataloaders_dict = {\"train\": train_data_loader, \"val\": valid_data_loader}\n    train_model(model, dataloaders_dict, criterion, optimizer, num_epochs=3)\n    \n    '''\n    best_score = 0\n    for epoch in range(Config.EPOCHS):\n        print(epoch)\n        train(train_data_loader, model, optimizer, device)\n        predictions, targets = evaluation(valid_data_loader, model, device)\n        F1_score = pfbeta(targets, predictions, beta=1)\n        if F1_score > best_score:\n            best_score = F1_score\n            torch.save(model.state_dict(), Config.MODEL_PATH)\n        print('best_score = ', best_score)     \n    '''\n    \n        \n        \n        \n        ","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:01.138226Z","iopub.execute_input":"2023-01-31T13:37:01.138589Z","iopub.status.idle":"2023-01-31T13:37:01.153042Z","shell.execute_reply.started":"2023-01-31T13:37:01.138556Z","shell.execute_reply":"2023-01-31T13:37:01.152058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed):\n    '''Sets the seed of the entire notebook so results are the same every time we run.\n    This is for REPRODUCIBILITY.'''\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    # When running on the CuDNN backend, two further options must be set\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    # Set a fixed value for the hash seed\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    print('> SEEDING DONE')\n    \nset_seed(seed=42)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:01.1547Z","iopub.execute_input":"2023-01-31T13:37:01.155103Z","iopub.status.idle":"2023-01-31T13:37:01.169982Z","shell.execute_reply.started":"2023-01-31T13:37:01.155003Z","shell.execute_reply":"2023-01-31T13:37:01.168937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(model, dataloaders_dict, criterion, optimizer, num_epochs):\n    best_acc = 0.0\n\n    for epoch in range(num_epochs):\n        model.cuda()\n        \n        \n        for phase in ['train', 'val']:\n            if phase == 'train':\n                model.train()\n            else:\n                model.eval()\n                \n            epoch_loss = 0.0\n            epoch_acc = 0\n            best_f1_score = 0.0\n            \n            dataloader = dataloaders_dict[phase]\n            predictions = []\n            targets = []\n            for item in tqdm(dataloader, leave=False):\n                images = item[0].cuda().float()\n                classes = item[1].cuda().float()\n\n                optimizer.zero_grad()\n                \n                with torch.set_grad_enabled(phase == 'train'):\n                    output = model(images).squeeze()\n                    #print('output size = ', output.shape)\n                    #print('target size = ', classes.shape)\n                    loss = criterion(output, classes)\n\n                    if phase == 'train':\n                        loss.backward()\n                        optimizer.step()\n\n                    epoch_loss += loss.item() * len(output)\n                    \n                    probs = output.sigmoid()\n                    threshold = 0.5\n                    predicted_vals = probs > threshold\n                    epoch_acc += torch.sum(predicted_vals == classes.data)\n                    \n                    predictions.append(output.sigmoid().cpu().detach().numpy())\n                    targets.append(classes.cpu().numpy())\n                    \n            predictions = np.concatenate(predictions) \n            targets = np.concatenate(targets)\n            \n            f1_score = pfbeta(targets, predictions, beta = 1.0)\n            if f1_score > best_f1_score :\n                best_f1_score = f1_score\n            print('f1_score = ', f1_score)        \n\n            data_size = len(dataloader.dataset)\n            epoch_loss = epoch_loss / data_size\n            epoch_acc = epoch_acc.double() / data_size\n\n            print(f'Epoch {epoch + 1}/{num_epochs} | {phase:^5} | Loss: {epoch_loss:.4f} | Acc: {epoch_acc:.4f}')\n            \n        if epoch_acc > best_acc:\n            traced = torch.jit.trace(model.cpu(), torch.rand(1, 1, 224, 224))\n            traced.save('model.pth')\n            best_acc = epoch_acc","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:01.171755Z","iopub.execute_input":"2023-01-31T13:37:01.172105Z","iopub.status.idle":"2023-01-31T13:37:01.186758Z","shell.execute_reply.started":"2023-01-31T13:37:01.172072Z","shell.execute_reply":"2023-01-31T13:37:01.185805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if __name__ == '__main__' :\n    \n    # Read training csv\n    dfx = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv').dropna(axis=0).reset_index(drop=True)\n\n\n    kfold = StratifiedGroupKFold(n_splits=5)\n\n    for fold, (train_idx, valid_idx) in enumerate(kfold.split(dfx, dfx['cancer'].values, dfx['patient_id'].values)):\n        print(f\"{'='*40} Fold: {fold} / 5 {'='*40}\")\n\n        df_train = dfx.loc[train_idx].reset_index(drop=True)\n        df_valid = dfx.loc[valid_idx].reset_index(drop=True)\n        run(df_train, df_valid)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T13:37:01.188349Z","iopub.execute_input":"2023-01-31T13:37:01.188727Z","iopub.status.idle":"2023-01-31T13:37:14.741571Z","shell.execute_reply.started":"2023-01-31T13:37:01.188695Z","shell.execute_reply":"2023-01-31T13:37:14.74023Z"},"trusted":true},"execution_count":null,"outputs":[]}]}