{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import scipy.misc\nimport imageio\nimport sys\nimport torchvision as tv\nimport gc\nimport numpy as np\nimport pandas as pd\nimport os\nimport pydicom\nimport cv2\nimport torchvision\nfrom torchvision import transforms\nfrom PIL import Image\nfrom torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader, random_split, TensorDataset, Dataset, WeightedRandomSampler\nimport torch\nimport glob\nimport seaborn as sns\nimport re\nfrom sklearn.model_selection import GroupKFold\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom pylab import rcParams\nfrom tqdm import tqdm\nimport albumentations\nimport matplotlib.pyplot as plt\nbatches = 10000\nDEVICE = 'cuda' if torch.cuda.is_available()else'cpu'\nif DEVICE == 'cuda':\n    BATCH_SIZE = 1000\nelse:\n    BATCH_SIZE = 500\nprint(f\"Using {DEVICE}device\")\nimage_size_seg = (128, 128, 128)\nmsk_size = image_size_seg[0]\nimage_size_cls = 384\nn_slice_per_c = 15\nn_ch = 5\nbatch_size_seg = 1\nnum_workers = 2\ndata_dir = '/kaggle/input/rsna-breast-cancer-detection'\ntrain_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_df = train_df.reset_index(drop=True)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-03T09:17:14.515245Z","iopub.execute_input":"2023-05-03T09:17:14.515889Z","iopub.status.idle":"2023-05-03T09:17:19.858798Z","shell.execute_reply.started":"2023-05-03T09:17:14.515788Z","shell.execute_reply":"2023-05-03T09:17:19.856766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df)\ntrain_df.isnull().sum()#检测每列中的缺失值的数量","metadata":{"execution":{"iopub.status.busy":"2023-05-03T09:17:40.552366Z","iopub.execute_input":"2023-05-03T09:17:40.552964Z","iopub.status.idle":"2023-05-03T09:17:40.600782Z","shell.execute_reply.started":"2023-05-03T09:17:40.552918Z","shell.execute_reply":"2023-05-03T09:17:40.599526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#创建特征列表\ncolumn_names = ['site_id','patient_id','image_id','laterality','view','age','cancer','biopsy','invasive','BIRADS','implant','density','difficult_negative_case']\n#舍弃带有缺失值的数据\ntrain_df1 = train_df.dropna(how = 'any')\ntrain_df1.shape\ntrain_df1.to_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv', header=None, index=None, sep=' ')\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T09:24:48.934774Z","iopub.execute_input":"2023-05-03T09:24:48.935239Z","iopub.status.idle":"2023-05-03T09:24:49.168214Z","shell.execute_reply.started":"2023-05-03T09:24:48.935202Z","shell.execute_reply":"2023-05-03T09:24:49.166561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_df = train_df.reset_index(drop=True)\nprint(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T09:19:58.120275Z","iopub.execute_input":"2023-05-03T09:19:58.120734Z","iopub.status.idle":"2023-05-03T09:19:58.213815Z","shell.execute_reply.started":"2023-05-03T09:19:58.120692Z","shell.execute_reply":"2023-05-03T09:19:58.212401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n#随机采样25%的数据用于测试，剩下的75%用于构建训练集合\nX_train,X_val,y_train,y_val = train_test_split(train_df1[column_names[1:13]],train_df1[column_names[6]],test_size=0.25,random_state=33)\n#查验训练样本的数量和类别分布\ny_train.value_counts()\n#查验测试样本的数量和类别分布\ny_val.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T09:17:48.911797Z","iopub.execute_input":"2023-05-03T09:17:48.912195Z","iopub.status.idle":"2023-05-03T09:17:48.943004Z","shell.execute_reply.started":"2023-05-03T09:17:48.912166Z","shell.execute_reply":"2023-05-03T09:17:48.941692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(10, 5))\n########## PLOTING CANCER ################\nsplot = sns.countplot(ax = axes[0], x = train_df1['cancer'])\n\ns = train_df1['cancer'].value_counts()\naxes[1].pie(s, autopct=\"%.1f%%\", labels = s.keys())\nfig.suptitle('Cancer distribution')","metadata":{"execution":{"iopub.status.busy":"2023-05-03T09:17:52.613487Z","iopub.execute_input":"2023-05-03T09:17:52.614149Z","iopub.status.idle":"2023-05-03T09:17:52.941595Z","shell.execute_reply.started":"2023-05-03T09:17:52.614116Z","shell.execute_reply":"2023-05-03T09:17:52.940312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.nn import functional as F\n\n\nclass RestNetBasicBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, stride):\n        super(RestNetBasicBlock, self).__init__()\n        self.conv1 = nn.Conv2d(in_channels, out_channels, kernel_size=3, stride=stride, padding=1)\n        self.bn1 = nn.BatchNorm2d(out_channels)\n        self.conv2 = nn.Conv2d(out_channels, out_channels, kernel_size=3, stride=stride, padding=1)\n        self.bn2 = nn.BatchNorm2d(out_channels)\n\n    def forward(self, x):\n        output = self.conv1(x)\n        output = F.relu(self.bn1(output))\n        output = self.conv2(output)\n        output = self.bn2(output)\n        return F.relu(x + output)\n\n\nclass RestNetDownBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, stride):\n        super(RestNetDownBlock, self).__init__()\n        self.conv1 = nn.Conv2d(in_channels, out_channels, kernel_size=3, stride=stride[0], padding=1)\n        self.bn1 = nn.BatchNorm2d(out_channels)\n        self.conv2 = nn.Conv2d(out_channels, out_channels, kernel_size=3, stride=stride[1], padding=1)\n        self.bn2 = nn.BatchNorm2d(out_channels)\n        self.extra = nn.Sequential(\n            nn.Conv2d(in_channels, out_channels, kernel_size=1, stride=stride[0], padding=0),\n            nn.BatchNorm2d(out_channels)\n        )\n\n    def forward(self, x):\n        extra_x = self.extra(x)\n        output = self.conv1(x)\n        out = F.relu(self.bn1(output))\n\n        out = self.conv2(out)\n        out = self.bn2(out)\n        return F.relu(extra_x + out)\n\n\nclass RestNet18(nn.Module):\n    def __init__(self):\n        super(RestNet18, self).__init__()\n        self.conv1 = nn.Conv2d(3, 64, kernel_size=7, stride=2, padding=3)\n        self.bn1 = nn.BatchNorm2d(64)\n        self.maxpool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)\n\n        self.layer1 = nn.Sequential(RestNetBasicBlock(64, 64, 1),\n                                    RestNetBasicBlock(64, 64, 1))\n\n        self.layer2 = nn.Sequential(RestNetDownBlock(64, 128, [2, 1]),\n                                    RestNetBasicBlock(128, 128, 1))\n\n        self.layer3 = nn.Sequential(RestNetDownBlock(128, 256, [2, 1]),\n                                    RestNetBasicBlock(256, 256, 1))\n\n        self.layer4 = nn.Sequential(RestNetDownBlock(256, 512, [2, 1]),\n                                    RestNetBasicBlock(512, 512, 1))\n\n        self.avgpool = nn.AdaptiveAvgPool2d(output_size=(1, 1))\n\n        self.fc = nn.Linear(512, 10)\n\n    def forward(self, x):\n        out = self.conv1(x)\n        out = self.layer1(out)\n        out = self.layer2(out)\n        out = self.layer3(out)\n        out = self.layer4(out)\n        out = self.avgpool(out)\n        out = out.reshape(x.shape[0], -1)\n        out = self.fc(out)\n        return out\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T09:17:56.769254Z","iopub.execute_input":"2023-05-03T09:17:56.769707Z","iopub.status.idle":"2023-05-03T09:17:56.790943Z","shell.execute_reply.started":"2023-05-03T09:17:56.76967Z","shell.execute_reply":"2023-05-03T09:17:56.789356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class RSNADataset(Dataset):\n\n    def __init__(self, annotations_file, img_dir, transform=None):\n        self.df = pd.read_csv(annotations_file)\n        self.img_dir = img_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n    \n\n\n    def __getitem__(self, ind):\n        \n        img_path = f\"{self.img_dir}/{self.df.iloc[ind].patient_id}_{self.df.iloc[ind].image_id}.png\"\n        img = Image.open(img_path).convert('RGB')\n        \n        label = self.df.iloc[ind].cancer\n        # there is no need to normalize data, it has already been normalized\n        if self.transform:\n            img = self.transform(img).to(torch.float32) \n        else:\n            default_transform = transforms.Compose([transforms.ToTensor()])\n            img = default_transform(img).to(torch.float32)\n            \n        #sample = {\"image\" : img, \"label\": label}\n        return img, label","metadata":{"execution":{"iopub.status.busy":"2023-05-03T09:18:00.791749Z","iopub.execute_input":"2023-05-03T09:18:00.792212Z","iopub.status.idle":"2023-05-03T09:18:00.801924Z","shell.execute_reply.started":"2023-05-03T09:18:00.792176Z","shell.execute_reply":"2023-05-03T09:18:00.800252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torchvision import transforms, datasets\ntrain_csv = '/kaggle/input/rsna-breast-cancer-detection/train.csv'\nimgs_dir = '/kaggle/input/rsnamamorgaphybreastcancerrecognition512x512'\naugmentator = transforms.Compose([\n    # input for augmentator is always PIL image\n    # transforms.ToPILImage(),\n    transforms.RandomHorizontalFlip(0.5),\n    transforms.RandomVerticalFlip(0.5),\n    transforms.RandomRotation(5),\n    transforms.ToTensor(), # return it as a tensor and transforms it to [0, 1]\n])\ndataset = RSNADataset(train_csv, imgs_dir, augmentator)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-05T08:30:55.738557Z","iopub.execute_input":"2023-05-05T08:30:55.739521Z","iopub.status.idle":"2023-05-05T08:30:58.12253Z","shell.execute_reply.started":"2023-05-05T08:30:55.739414Z","shell.execute_reply":"2023-05-05T08:30:58.120795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use torch.utils.data to create a DataLoader \n# that will take care of creating batches \n\n# TODO, remove using half of dataset\n# dataset, _ = random_split(dataset, [int(len(dataset)*0.02), int(len(dataset)*0.98 + 1)])\n# split training into validation and train\nval_pct = 0.1\nval_size = int(val_pct * len(dataset))\ntrain_size = len(dataset) - val_size\ntrain_dataset, val_dataset = random_split(dataset, [train_size, val_size])","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:40:40.59035Z","iopub.execute_input":"2023-02-28T07:40:40.590667Z","iopub.status.idle":"2023-02-28T07:40:40.623168Z","shell.execute_reply.started":"2023-02-28T07:40:40.590638Z","shell.execute_reply":"2023-02-28T07:40:40.622439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 64\n\n# Applying random sampler just tu train dataset, not for validation, since the validation dataset should be imitation of 'real' DS\ntrain_dataloader = torch.utils.data.DataLoader(train_dataset, batch_size=batch_size,shuffle = False,num_workers = 0)\nval_dataloader = torch.utils.data.DataLoader(val_dataset, batch_size=batch_size, shuffle = False,num_workers = 0)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:40:40.624334Z","iopub.execute_input":"2023-02-28T07:40:40.62512Z","iopub.status.idle":"2023-02-28T07:40:40.631824Z","shell.execute_reply.started":"2023-02-28T07:40:40.625081Z","shell.execute_reply":"2023-02-28T07:40:40.630453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataloaders = {'train' : train_dataloader, 'val' : val_dataloader}\ndataset_sizes = {'train': train_size, 'val' : val_size}\nprint(len(train_dataset), len(val_dataset))\nprint(len(train_dataloader), len(val_dataloader))","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:40:40.633448Z","iopub.execute_input":"2023-02-28T07:40:40.633934Z","iopub.status.idle":"2023-02-28T07:40:40.65062Z","shell.execute_reply.started":"2023-02-28T07:40:40.633898Z","shell.execute_reply":"2023-02-28T07:40:40.648357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rows = 1\ncols = 4\nplt.subplots(rows, cols, figsize = (20, 20))\n\nbatch_imgs, batch_labels = next(iter(train_dataloader))\ni = 0\nfor img in batch_imgs:\n    if i >= rows*cols:\n        break\n    plt.subplot(rows, cols, i+1)\n    plt.title(\"Cancer\" if batch_labels[i] == 1 else \"No cancer\")\n    plt.imshow(img.permute(1, 2, 0))\n\n    i += 1\n\nlabels_count = np.zeros(2)\nfor l in batch_labels:\n    labels_count[l] += 1 \n    \nprint(f'There are {labels_count[0]} negative and {labels_count[1]} positive samples in this batch.')","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:40:40.652906Z","iopub.execute_input":"2023-02-28T07:40:40.653594Z","iopub.status.idle":"2023-02-28T07:40:44.226986Z","shell.execute_reply.started":"2023-02-28T07:40:40.653558Z","shell.execute_reply":"2023-02-28T07:40:44.226156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"net = RestNet18()\nnet.to(DEVICE)\nloss_function = nn.CrossEntropyLoss()\noptimizer = optim.Adam(net.parameters(), lr=0.0001)\n\nbest_acc = 0.0\nsave_path = '/kaggle/working/RestNet18.pth'","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:40:44.230778Z","iopub.execute_input":"2023-02-28T07:40:44.231071Z","iopub.status.idle":"2023-02-28T07:40:44.514787Z","shell.execute_reply.started":"2023-02-28T07:40:44.231043Z","shell.execute_reply":"2023-02-28T07:40:44.513279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_num = len(val_dataset)\ntrain_steps = len(train_dataloader)\n\nfor epoch in range(2):\n    # train\n    net.train()\n    running_loss = 0.0\n    for step, data in enumerate(train_dataloader):\n        images, labels = data\n        optimizer.zero_grad()\n        logits, aux_logits2, aux_logits1 = net(images.to(DEVICE))\n        loss0 = loss_function(logits, labels.to(DEVICE))\n        loss1 = loss_function(aux_logits1, labels.to(DEVICE))\n        loss2 = loss_function(aux_logits2, labels.to(DEVICE))\n        loss = loss0 + loss1 * 0.3 + loss2 * 0.3\n        loss.backward()\n        optimizer.step()\n\n            # print statistics\n        running_loss += loss.item()\n\n        train_dataloader.desc = \"train epoch[{}/{}] loss:{:.3f}\".format(epoch + 1,\n                                                                     2,\n                                                                     loss)\n\n        # validate\n    net.eval()\n    acc = 0.0  # accumulate accurate number / epoch\n    with torch.no_grad():\n        val_bar = tqdm(val_dataloader)\n        for val_data in val_bar:\n            val_images, val_labels = val_data\n            outputs = net(val_images.to(DEVICE))  # eval model only have last output layer\n            predict_y = torch.max(outputs, dim=1)[1]\n            acc += torch.eq(predict_y, val_labels.to(DEVICE)).sum().item()\n\n    val_accurate = acc / val_num\n    print('[epoch %d] train_loss: %.3f  val_accuracy: %.3f' %\n            (epoch + 1, running_loss / train_steps, val_accurate))\n\n    if val_accurate > best_acc:\n        best_acc = val_accurate\n        torch.save(net.state_dict(), save_path)\nprint('Finished Training')\n","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:40:44.516467Z","iopub.execute_input":"2023-02-28T07:40:44.516802Z","iopub.status.idle":"2023-02-28T07:41:03.458723Z","shell.execute_reply.started":"2023-02-28T07:40:44.516773Z","shell.execute_reply":"2023-02-28T07:41:03.456879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"net = RestNet18(num_classes=2, aux_logits=True, init_weights=True)\n# load model weights\nmodel_weight_path = \"/kaggle/input/goolenetmodel/RestNet18.pth\"\nmodel_weight = torch.load(model_weight_path, map_location='cpu')\nnet.load_state_dict(model_weight,False)\nnet.eval()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:41:03.459799Z","iopub.status.idle":"2023-02-28T07:41:03.460239Z","shell.execute_reply.started":"2023-02-28T07:41:03.460003Z","shell.execute_reply":"2023-02-28T07:41:03.460023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = RestNet18()\nmodel.to(DEVICE)\nmodel.eval()\n''''y_pred = []\ny_true = []\nfor inputs, labels in val_dataloader:\n        outputs = model(inputs) # Feed Network\n        outputs = torch.sigmoid(outputs)\n        \n        for i in range(len(outputs)):\n            output = outputs[i][0] \n            output = output.float()\n            if output > 0.5:\n                y_pred.append(int(1)) # Save Prediction\n            else:\n                y_pred.append(int(0)) # Save Prediction\n        print('1')\n        labels = labels.data.cpu().numpy()\n        y_true.extend(labels) # Save Truth\n\n# constant for classes\nclasses = ('Not Malignant', 'Malignant Cancer')\n\n# Build confusion matrix\n#cm = confusion_matrix(y_true, y_pred)'''","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:41:03.462279Z","iopub.status.idle":"2023-02-28T07:41:03.462722Z","shell.execute_reply.started":"2023-02-28T07:41:03.462484Z","shell.execute_reply":"2023-02-28T07:41:03.462505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dir = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntest_dir.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:41:03.463818Z","iopub.status.idle":"2023-02-28T07:41:03.4643Z","shell.execute_reply.started":"2023-02-28T07:41:03.464025Z","shell.execute_reply":"2023-02-28T07:41:03.464047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class testDataset(Dataset):\n\n    def __init__(self, annotations_file, img_dir, transform=None):\n        self.df = pd.read_csv(annotations_file)\n        self.img_dir = img_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n    \n\n\n    def __getitem__(self, ind):\n        \n        img_path = f\"{self.img_dir}/{self.df.iloc[ind].patient_id}_{self.df.iloc[ind].image_id}.png\"\n        img = Image.open(img_path).convert('RGB')\n        \n        #label = self.df.iloc[ind].cancer\n        # there is no need to normalize data, it has already been normalized\n        if self.transform:\n            img = self.transform(img).to(torch.float32) \n        else:\n            default_transform = transforms.Compose([transforms.ToTensor()])\n            img = default_transform(img).to(torch.float32)\n            \n        #sample = {\"image\" : img, \"label\": label}\n        return img#, label\n\n\n\naugmentator = transforms.Compose([\n    # input for augmentator is always PIL image\n    # transforms.ToPILImage(),\n    transforms.RandomHorizontalFlip(0.5),\n    transforms.RandomVerticalFlip(0.5),\n    transforms.RandomRotation(5),\n    transforms.ToTensor(), # return it as a tensor and transforms it to [0, 1]\n])\ntest_image = '/kaggle/input/test-image'\ntest_csv = '/kaggle/input/rsna-breast-cancer-detection/test.csv'\ntestset = testDataset(test_csv, test_image, augmentator)\ntest_dataloader = torch.utils.data.DataLoader(testset, batch_size=12,shuffle = False,num_workers = 0)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:41:03.466737Z","iopub.status.idle":"2023-02-28T07:41:03.467354Z","shell.execute_reply.started":"2023-02-28T07:41:03.467115Z","shell.execute_reply":"2023-02-28T07:41:03.467137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = RestNet18()\nmodel.to(DEVICE)\nmodel.eval()\ncancer = []\nfor inputs in test_dataloader:\n        outputs = model(inputs) # Feed Network\n        outputs = torch.sigmoid(outputs)\n        \n        for i in range(len(outputs)):\n            output = outputs[i][0] \n            output = output.float()\n            cancer.append(output.item())\ncancer","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:41:03.468448Z","iopub.status.idle":"2023-02-28T07:41:03.469025Z","shell.execute_reply.started":"2023-02-28T07:41:03.468817Z","shell.execute_reply":"2023-02-28T07:41:03.468838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:41:03.470141Z","iopub.status.idle":"2023-02-28T07:41:03.470737Z","shell.execute_reply.started":"2023-02-28T07:41:03.470517Z","shell.execute_reply":"2023-02-28T07:41:03.470539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['cancer'] = cancer\ndf_test","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:41:03.471808Z","iopub.status.idle":"2023-02-28T07:41:03.472408Z","shell.execute_reply.started":"2023-02-28T07:41:03.472162Z","shell.execute_reply":"2023-02-28T07:41:03.472183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = df_test.loc[:, 'prediction_id':'cancer']\nsubmission","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:41:03.473504Z","iopub.status.idle":"2023-02-28T07:41:03.47408Z","shell.execute_reply.started":"2023-02-28T07:41:03.473872Z","shell.execute_reply":"2023-02-28T07:41:03.473892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = submission.groupby('prediction_id').mean().reset_index()\nsubmission","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:41:03.475135Z","iopub.status.idle":"2023-02-28T07:41:03.475745Z","shell.execute_reply.started":"2023-02-28T07:41:03.475525Z","shell.execute_reply":"2023-02-28T07:41:03.475554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(data = {'prediction_id': df_test['prediction_id'], 'cancer': cancer})\nsubmission","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:41:03.477171Z","iopub.status.idle":"2023-02-28T07:41:03.477729Z","shell.execute_reply.started":"2023-02-28T07:41:03.477485Z","shell.execute_reply":"2023-02-28T07:41:03.477506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = submission.groupby('prediction_id').mean().reset_index()\nsubmission","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:41:03.479323Z","iopub.status.idle":"2023-02-28T07:41:03.479684Z","shell.execute_reply.started":"2023-02-28T07:41:03.47951Z","shell.execute_reply":"2023-02-28T07:41:03.479526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T07:41:03.481036Z","iopub.status.idle":"2023-02-28T07:41:03.481407Z","shell.execute_reply.started":"2023-02-28T07:41:03.481233Z","shell.execute_reply":"2023-02-28T07:41:03.48125Z"},"trusted":true},"execution_count":null,"outputs":[]}]}