{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","scrolled":true,"execution":{"iopub.status.busy":"2022-12-01T08:13:19.774866Z","iopub.execute_input":"2022-12-01T08:13:19.775572Z","iopub.status.idle":"2022-12-01T08:13:19.797096Z","shell.execute_reply.started":"2022-12-01T08:13:19.775464Z","shell.execute_reply":"2022-12-01T08:13:19.796169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Libraries**","metadata":{}},{"cell_type":"code","source":"!pip3 install python-gdcm\n!pip3 install pylibjpeg pylibjpeg-libjpeg","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:21.22876Z","iopub.execute_input":"2022-12-01T08:13:21.229355Z","iopub.status.idle":"2022-12-01T08:13:45.000212Z","shell.execute_reply.started":"2022-12-01T08:13:21.229307Z","shell.execute_reply":"2022-12-01T08:13:44.998998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch import nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms, models\n\nimport pandas as pd\nimport numpy as np\nimport cv2\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import precision_score, recall_score\n\nimport pydicom\nimport gdcm\nimport pylibjpeg\n\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:45.002831Z","iopub.execute_input":"2022-12-01T08:13:45.003802Z","iopub.status.idle":"2022-12-01T08:13:47.630812Z","shell.execute_reply.started":"2022-12-01T08:13:45.003767Z","shell.execute_reply":"2022-12-01T08:13:47.629759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(1)","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:47.632316Z","iopub.execute_input":"2022-12-01T08:13:47.632952Z","iopub.status.idle":"2022-12-01T08:13:47.639115Z","shell.execute_reply.started":"2022-12-01T08:13:47.632913Z","shell.execute_reply":"2022-12-01T08:13:47.637884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Working with data**\nThis module is about data, EDA, Dataset and Dataloader, visualization","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ndf","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:47.641484Z","iopub.execute_input":"2022-12-01T08:13:47.641853Z","iopub.status.idle":"2022-12-01T08:13:47.774942Z","shell.execute_reply.started":"2022-12-01T08:13:47.641816Z","shell.execute_reply":"2022-12-01T08:13:47.774026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"patient_id\"].nunique(), df.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:49.486253Z","iopub.execute_input":"2022-12-01T08:13:49.486622Z","iopub.status.idle":"2022-12-01T08:13:49.502926Z","shell.execute_reply.started":"2022-12-01T08:13:49.486591Z","shell.execute_reply":"2022-12-01T08:13:49.501751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cheking how many images and patients in the data","metadata":{}},{"cell_type":"code","source":"sns.histplot(data=df, x=\"cancer\")","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:50.196757Z","iopub.execute_input":"2022-12-01T08:13:50.197117Z","iopub.status.idle":"2022-12-01T08:13:50.458108Z","shell.execute_reply.started":"2022-12-01T08:13:50.197084Z","shell.execute_reply":"2022-12-01T08:13:50.457174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There we can see that classes a very imbalanced, there are a little negaitve class data","metadata":{}},{"cell_type":"code","source":"sns.histplot(data=df[df[\"cancer\"] == 0], x=\"age\")","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:50.501442Z","iopub.execute_input":"2022-12-01T08:13:50.502344Z","iopub.status.idle":"2022-12-01T08:13:50.844148Z","shell.execute_reply.started":"2022-12-01T08:13:50.502296Z","shell.execute_reply":"2022-12-01T08:13:50.843104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(data=df[df[\"cancer\"] == 1], x=\"age\")","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:50.887432Z","iopub.execute_input":"2022-12-01T08:13:50.887798Z","iopub.status.idle":"2022-12-01T08:13:51.126Z","shell.execute_reply.started":"2022-12-01T08:13:50.887766Z","shell.execute_reply":"2022-12-01T08:13:51.125046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There I cheked age distribution of each class and as we can see cancer doesn't depend on age","metadata":{}},{"cell_type":"code","source":"def visualize(patient_id, figsize=(20, 20)):\n    fig, ax = plt.subplots(2, 2, figsize=figsize)\n    ax = ax.flatten()\n    path_to_dcms = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\n    for i, dcm_path in enumerate(os.listdir(os.path.join(path_to_dcms, patient_id))):\n        dcm = pydicom.dcmread(os.path.join(path_to_dcms, patient_id, dcm_path))\n        dcm = dcm.pixel_array\n        dcm = (dcm - dcm.min()) / (dcm.max() - dcm.min()) * 255\n        ax[i].imshow(dcm, cmap=\"bone\")","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:51.526036Z","iopub.execute_input":"2022-12-01T08:13:51.526743Z","iopub.status.idle":"2022-12-01T08:13:51.533619Z","shell.execute_reply.started":"2022-12-01T08:13:51.526706Z","shell.execute_reply":"2022-12-01T08:13:51.532496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualize(\"10011\", figsize=(10, 10))","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:51.859036Z","iopub.execute_input":"2022-12-01T08:13:51.860148Z","iopub.status.idle":"2022-12-01T08:13:56.82999Z","shell.execute_reply.started":"2022-12-01T08:13:51.860101Z","shell.execute_reply":"2022-12-01T08:13:56.829016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_ids = df[\"patient_id\"].unique()\nnp.random.shuffle(patient_ids)\n\ntrain_size = int(len(patient_ids) * 0.8)\n\ntrain_ids = patient_ids[:train_size]\nval_ids = patient_ids[train_size:]\nf\"Train: {len(train_ids)} Val: {len(val_ids)}\"","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:57.908926Z","iopub.execute_input":"2022-12-01T08:13:57.909342Z","iopub.status.idle":"2022-12-01T08:13:57.921483Z","shell.execute_reply.started":"2022-12-01T08:13:57.909305Z","shell.execute_reply":"2022-12-01T08:13:57.920232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_to_drop = [\"laterality\", \"view\", \"age\", \n                   \"biopsy\", \"invasive\", \"BIRADS\", \n                   \"implant\", \"density\", \"machine_id\",\n                   \"difficult_negative_case\", \"site_id\"]\n\ntrain_df = df[df[\"patient_id\"].isin(train_ids)].drop(columns_to_drop, axis=1).reset_index()\nval_df = df[df[\"patient_id\"].isin(val_ids)].drop(columns_to_drop, axis=1).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:58.271453Z","iopub.execute_input":"2022-12-01T08:13:58.271819Z","iopub.status.idle":"2022-12-01T08:13:58.290389Z","shell.execute_reply.started":"2022-12-01T08:13:58.271788Z","shell.execute_reply":"2022-12-01T08:13:58.289472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"cancer\"].value_counts(), val_df[\"cancer\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:58.594494Z","iopub.execute_input":"2022-12-01T08:13:58.594865Z","iopub.status.idle":"2022-12-01T08:13:58.605432Z","shell.execute_reply.started":"2022-12-01T08:13:58.594831Z","shell.execute_reply":"2022-12-01T08:13:58.604471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There triain test split is created. There aren-t same patients in train in test split","metadata":{}},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:13:59.865274Z","iopub.execute_input":"2022-12-01T08:13:59.865632Z","iopub.status.idle":"2022-12-01T08:13:59.879042Z","shell.execute_reply.started":"2022-12-01T08:13:59.865601Z","shell.execute_reply":"2022-12-01T08:13:59.877888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CancerDataset(Dataset):\n    def __init__(self, df, transform=None):\n        super(CancerDataset, self).__init__()\n        self.df = df.copy()\n        self.transform = transform\n        \n        self.path_to_dcms = \"/kaggle/input/rsna-breast-cancer-256-pngs\"\n        \n    def __getitem__(self, idx):\n        img_path = os.path.join(self.path_to_dcms, f\"{self.df.loc[idx, 'patient_id']}_{self.df.loc[idx, 'image_id']}.png\")\n        img = cv2.imread(img_path)\n        if self.transform:\n            img = self.transform(img)\n        label = self.df.loc[idx, \"cancer\"]\n        label = torch.tensor(label, dtype=torch.float32).unsqueeze(0)\n        return img, label\n    \n    def __len__(self):\n        return len(self.df)","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:14:00.692771Z","iopub.execute_input":"2022-12-01T08:14:00.69315Z","iopub.status.idle":"2022-12-01T08:14:00.700918Z","shell.execute_reply.started":"2022-12-01T08:14:00.693114Z","shell.execute_reply":"2022-12-01T08:14:00.699753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dataset class, using PyTorch","metadata":{}},{"cell_type":"code","source":"train_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.ColorJitter(brightness=(0.5, 1.5)),\n    transforms.Resize((256, 256))\n])\nval_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Resize((256, 256))\n])\n\ntrain_dataset = CancerDataset(train_df, transform=train_transform)\nval_dataset = CancerDataset(val_df, transform=val_transform)","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:14:01.309668Z","iopub.execute_input":"2022-12-01T08:14:01.310033Z","iopub.status.idle":"2022-12-01T08:14:01.319117Z","shell.execute_reply.started":"2022-12-01T08:14:01.310002Z","shell.execute_reply":"2022-12-01T08:14:01.31781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here some augmentations are defined. The images are resized to 256x256 for resnet18","metadata":{"execution":{"iopub.status.busy":"2022-11-30T09:22:37.654327Z","iopub.execute_input":"2022-11-30T09:22:37.654715Z","iopub.status.idle":"2022-11-30T09:22:37.662617Z","shell.execute_reply.started":"2022-11-30T09:22:37.654682Z","shell.execute_reply":"2022-11-30T09:22:37.660955Z"}}},{"cell_type":"code","source":"img, label = train_dataset[0]\nplt.imshow(img.permute(1, 2, 0).numpy(), cmap=\"bone\")","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:14:02.13179Z","iopub.execute_input":"2022-12-01T08:14:02.134193Z","iopub.status.idle":"2022-12-01T08:14:02.384026Z","shell.execute_reply.started":"2022-12-01T08:14:02.134134Z","shell.execute_reply":"2022-12-01T08:14:02.383126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Model**\nIn this module I'll create model based on ResNet18","metadata":{}},{"cell_type":"code","source":"class ResNetModel(nn.Module):\n    def __init__(self):\n        super(ResNetModel, self).__init__()\n        backbone = models.resnet50(pretrained=True)\n        backbone.fc = nn.Linear(2048, 1)\n        self.backbone = backbone\n        \n    def forward(self, dcm):\n        out = self.backbone(dcm)\n        out = torch.sigmoid(out)\n        return out","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:15:36.102713Z","iopub.execute_input":"2022-12-01T08:15:36.103089Z","iopub.status.idle":"2022-12-01T08:15:36.110097Z","shell.execute_reply.started":"2022-12-01T08:15:36.10306Z","shell.execute_reply":"2022-12-01T08:15:36.10898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = ResNetModel()\nparams = sum(p.numel() for p in model.parameters())\ndel model\nf\"Num parameteres: {round(params / 1000000, 1)} m.\"","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:14:15.038777Z","iopub.execute_input":"2022-12-01T08:14:15.03915Z","iopub.status.idle":"2022-12-01T08:14:21.149863Z","shell.execute_reply.started":"2022-12-01T08:14:15.039116Z","shell.execute_reply":"2022-12-01T08:14:21.148791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Train pipeline**\nIn this module I'll desribe training loop","metadata":{}},{"cell_type":"code","source":"DEVICE = \"cuda:0\" if torch.cuda.is_available() else \"cpu\"\nDEVICE","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:14:23.710305Z","iopub.execute_input":"2022-12-01T08:14:23.710678Z","iopub.status.idle":"2022-12-01T08:14:23.78289Z","shell.execute_reply.started":"2022-12-01T08:14:23.710645Z","shell.execute_reply":"2022-12-01T08:14:23.781586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_batch(batch, model, criterion, optimizer, threshold=0.5):\n    optimizer.zero_grad()\n    \n    dcms, labels = batch\n    dcms, labels = dcms.to(DEVICE), labels.to(DEVICE)\n    \n    preds = model(dcms)\n\n    precision = precision_score(labels.detach().cpu().numpy(), preds.detach().cpu().numpy() > threshold, zero_division=0)\n    recall = recall_score(labels.detach().cpu().numpy(), preds.detach().cpu().numpy() > threshold, zero_division=0)\n    \n    loss = criterion(preds, labels)\n    loss.backward()\n\n    optimizer.step()\n    return loss, precision, recall","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:14:25.165079Z","iopub.execute_input":"2022-12-01T08:14:25.165781Z","iopub.status.idle":"2022-12-01T08:14:25.173081Z","shell.execute_reply.started":"2022-12-01T08:14:25.165744Z","shell.execute_reply":"2022-12-01T08:14:25.171756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def eval_batch(batch, model, criterion, threshold=0.5):\n    dcms, labels = batch\n    dcms, labels = dcms.to(DEVICE), labels.to(DEVICE)\n    \n    preds = model(dcms)\n    \n    precision = precision_score(labels.detach().cpu().numpy(), preds.detach().cpu().numpy() > threshold, zero_division=0)\n    recall = recall_score(labels.detach().cpu().numpy(), preds.detach().cpu().numpy() > threshold, zero_division=0)\n    \n    loss = criterion(preds, labels)\n    return loss, precision, recall","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:14:26.79018Z","iopub.execute_input":"2022-12-01T08:14:26.791336Z","iopub.status.idle":"2022-12-01T08:14:26.799495Z","shell.execute_reply.started":"2022-12-01T08:14:26.791286Z","shell.execute_reply":"2022-12-01T08:14:26.79817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_loop(train_dataset, val_dataset, n_epoch, batch_size=16, num_workers=8, lr=1e-4):\n    model = ResNetModel().to(DEVICE)\n    criterion = nn.BCELoss(reduction=\"mean\")\n    optimizer = torch.optim.Adam(model.parameters(), lr=lr)\n    \n    train_dataloader = DataLoader(train_dataset, batch_size=batch_size, num_workers=num_workers, shuffle=True)\n    val_dataloader = DataLoader(val_dataset, batch_size=batch_size, num_workers=num_workers, shuffle=False)\n    \n    for epoch in range(1, n_epoch + 1):\n        print(f\"Epoch {epoch}\\n\", \"-\" * 50)\n        \n        train_loss = 0\n        train_precision = 0\n        train_recall = 0\n        \n        val_loss = 0\n        val_precision = 0\n        val_recall = 0\n        \n        model.train()\n        for batch in train_dataloader:\n            loss, precision, recall = train_batch(batch, model, criterion, optimizer)\n            \n            with torch.no_grad():\n                train_loss += loss\n                train_precision += precision\n                train_recall += recall\n            \n        \n        print(f\"Train Loss: {train_loss / len(train_dataloader)} Precision: {train_precision / len(train_dataloader)} Recall: {train_recall / len(train_dataloader)}\")\n        \n        model.eval()\n        for batch in val_dataloader:\n            with torch.no_grad():\n                loss, precision, recall = eval_batch(batch, model, criterion)\n                val_loss += loss\n                val_precision += precision\n                val_recall += recall\n                \n        print(f\"Val Loss: {val_loss / len(val_dataloader)} Precision: {val_precision / len(val_dataloader)} Recall: {val_recall / len(val_dataloader)}\")\n        torch.save(model.state_dict(), f\"res50_ep{epoch}.pth\")\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:15:11.228318Z","iopub.execute_input":"2022-12-01T08:15:11.228689Z","iopub.status.idle":"2022-12-01T08:15:11.23977Z","shell.execute_reply.started":"2022-12-01T08:15:11.228659Z","shell.execute_reply":"2022-12-01T08:15:11.238528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_epoch = 100\nbatch_size = 32\nnum_workers = 2\nlr = 1e-4\n\nmodel = train_loop(train_dataset, val_dataset, n_epoch=n_epoch, batch_size=batch_size, num_workers=num_workers, lr=lr)","metadata":{"execution":{"iopub.status.busy":"2022-12-01T08:15:39.720817Z","iopub.execute_input":"2022-12-01T08:15:39.721205Z","iopub.status.idle":"2022-12-01T10:24:04.095766Z","shell.execute_reply.started":"2022-12-01T08:15:39.721147Z","shell.execute_reply":"2022-12-01T10:24:04.093982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv\")\nsample_submission","metadata":{"execution":{"iopub.status.busy":"2022-12-01T10:24:18.106874Z","iopub.execute_input":"2022-12-01T10:24:18.107363Z","iopub.status.idle":"2022-12-01T10:24:18.132824Z","shell.execute_reply.started":"2022-12-01T10:24:18.107322Z","shell.execute_reply":"2022-12-01T10:24:18.131906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-12-01T10:24:20.403148Z","iopub.execute_input":"2022-12-01T10:24:20.403639Z","iopub.status.idle":"2022-12-01T10:24:20.435745Z","shell.execute_reply.started":"2022-12-01T10:24:20.403601Z","shell.execute_reply":"2022-12-01T10:24:20.434777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_to_dcms = \"/kaggle/input/rsna-breast-cancer-detection/test_images\"\npreds = []\n\nmodel = ResNetModel().to(DEVICE)\nmodel.load_state_dict(torch.load(\"res50_ep18.pth\"))\nmodel.eval()\n\nfor i in range(len(test_df)):\n    dcm_path = os.path.join(path_to_dcms, str(test_df.loc[i, \"patient_id\"]), f\"{test_df.loc[i, 'image_id']}.dcm\")\n    dcm = pydicom.dcmread(dcm_path)\n    dcm = dcm.pixel_array\n    dcm = (dcm - dcm.min()) / (dcm.max() - dcm.min())\n    dcm = torch.FloatTensor(dcm)\n    dcm = torch.stack([dcm, dcm, dcm], axis=0)\n    dcm = transforms.Resize((256, 256))(dcm)\n    dcm = dcm.unsqueeze(0).to(DEVICE)\n    pred = model(dcm).detach().cpu().numpy()[0].item()\n    preds.append(pred)\n    \npreds","metadata":{"execution":{"iopub.status.busy":"2022-12-01T10:24:44.707046Z","iopub.execute_input":"2022-12-01T10:24:44.707525Z","iopub.status.idle":"2022-12-01T10:24:48.788103Z","shell.execute_reply.started":"2022-12-01T10:24:44.707491Z","shell.execute_reply":"2022-12-01T10:24:48.7872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_data = [np.mean(preds[:2]), np.mean(preds[2:])]","metadata":{"execution":{"iopub.status.busy":"2022-12-01T10:24:50.912737Z","iopub.execute_input":"2022-12-01T10:24:50.913624Z","iopub.status.idle":"2022-12-01T10:24:50.919308Z","shell.execute_reply.started":"2022-12-01T10:24:50.913576Z","shell.execute_reply":"2022-12-01T10:24:50.918331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission[\"cancer\"] = sub_data\nsample_submission","metadata":{"execution":{"iopub.status.busy":"2022-12-01T10:24:52.246682Z","iopub.execute_input":"2022-12-01T10:24:52.247044Z","iopub.status.idle":"2022-12-01T10:24:52.257596Z","shell.execute_reply.started":"2022-12-01T10:24:52.247013Z","shell.execute_reply":"2022-12-01T10:24:52.256656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-01T10:24:57.780183Z","iopub.execute_input":"2022-12-01T10:24:57.781117Z","iopub.status.idle":"2022-12-01T10:24:57.791697Z","shell.execute_reply.started":"2022-12-01T10:24:57.781071Z","shell.execute_reply":"2022-12-01T10:24:57.79054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}