{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Load the requirement files","metadata":{}},{"cell_type":"markdown","source":"Note: The files shown below can be downloaded from the outputs section of the kaggle notebook.","metadata":{}},{"cell_type":"code","source":"pip install /kaggle/input/my-requirement-files/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl # installing dicom library","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:49:18.410629Z","iopub.execute_input":"2023-03-20T13:49:18.411143Z","iopub.status.idle":"2023-03-20T13:49:50.846481Z","shell.execute_reply.started":"2023-03-20T13:49:18.411094Z","shell.execute_reply":"2023-03-20T13:49:50.845374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# code below is used to load the EfficientNet library\nimport sys\nsys.path.append('/kaggle/input/library-load/efficientnet_pytorch-0.7.1') ","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:49:50.848625Z","iopub.execute_input":"2023-03-20T13:49:50.848979Z","iopub.status.idle":"2023-03-20T13:49:50.855361Z","shell.execute_reply.started":"2023-03-20T13:49:50.848943Z","shell.execute_reply":"2023-03-20T13:49:50.853592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2 # for resizing images\nimport torch # for pytorch framework\nimport dicomsdl # for reading dicom files\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns # for visualization\nfrom PIL import Image # to preprocess and transform images \nimport torch.nn as nn # for defining custom neural network modules\nimport tensorflow as tf # tensorflow framework\nimport torch.optim as optim # for defining our optimizer\nimport matplotlib.pyplot as plt\nfrom torchvision import transforms # for image transformation and augmentation\nfrom torch.utils.data import Dataset # to create custom dataset in pytorch \nfrom torch.utils.data import DataLoader # provides an interface for batching and iterating over a dataset object\nfrom efficientnet_pytorch import EfficientNet # for loading the efficient net pre-trained model\nfrom torchvision.transforms import ToPILImage # for visualizing output of neural network\nfrom sklearn.model_selection import train_test_split # create training and validation sets\n\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:49:50.856649Z","iopub.execute_input":"2023-03-20T13:49:50.857323Z","iopub.status.idle":"2023-03-20T13:50:02.030291Z","shell.execute_reply.started":"2023-03-20T13:49:50.857287Z","shell.execute_reply":"2023-03-20T13:50:02.029184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Global Variable Declaration","metadata":{}},{"cell_type":"code","source":"# image re-size dimensions\ntarget_size = [224,224]\n\n# batch size for train data\nbatch_size = 16 \n# batch size for test data\ntest_batch_size = 32 \n\n# number of epochs\nnum_epochs = 6 ","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:02.031692Z","iopub.execute_input":"2023-03-20T13:50:02.032452Z","iopub.status.idle":"2023-03-20T13:50:02.037398Z","shell.execute_reply.started":"2023-03-20T13:50:02.03242Z","shell.execute_reply":"2023-03-20T13:50:02.036367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read Data","metadata":{}},{"cell_type":"code","source":"data_dir = '/kaggle/input/rsna-breast-cancer-detection'","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:02.043457Z","iopub.execute_input":"2023-03-20T13:50:02.0441Z","iopub.status.idle":"2023-03-20T13:50:02.062365Z","shell.execute_reply.started":"2023-03-20T13:50:02.044063Z","shell.execute_reply":"2023-03-20T13:50:02.061158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(f\"{data_dir}/train.csv\")\n\ntrain_df['dcm_path'] = train_df.apply(\n    lambda i: os.path.join(\n        f\"{data_dir}\", 'train_images', str(i['patient_id']), str(i['image_id']) + '.dcm'\n    ), axis=1\n)\nprint(train_df.shape)\ntrain_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:02.063915Z","iopub.execute_input":"2023-03-20T13:50:02.064355Z","iopub.status.idle":"2023-03-20T13:50:03.048113Z","shell.execute_reply.started":"2023-03-20T13:50:02.064319Z","shell.execute_reply":"2023-03-20T13:50:03.046935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(f\"{data_dir}/test.csv\")\n\ntest_df['dcm_path'] = test_df.apply(\n    lambda i: os.path.join(\n        f\"{data_dir}\", 'test_images', str(i['patient_id']), str(i['image_id']) + '.dcm'\n    ), axis=1\n)\nprint(test_df.shape)\ntest_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:03.049827Z","iopub.execute_input":"2023-03-20T13:50:03.050539Z","iopub.status.idle":"2023-03-20T13:50:03.076047Z","shell.execute_reply.started":"2023-03-20T13:50:03.050493Z","shell.execute_reply":"2023-03-20T13:50:03.074979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"## Metadata for each patient and image\n**site_id** - ID code for the source hospital.\n\n**patient_id** - ID code for the patient.\n\n**image_id** - ID code for the image.\n\n**laterality** - Whether the image is of the left or right breast.\n\n**view**  - The orientation of the image. The default for a screening exam is to capture two views per breast.\n\n**age** - The patient's age in years.\n\n**implant** - Whether or not the patient had breast implants. Site 1 only provides breast implant information at the patient level, not at the breast level.\n\n**density** - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult. Only provided for train.\n\n**machine_id** - An ID code for the imaging device.\n\n**cancer** - Whether or not the breast was positive for malignant cancer. The target value. Only provided for train.\n\n**biopsy** - Whether or not a follow-up biopsy was performed on the breast. Only provided for train.\n\n**invasive** - If the breast is positive for cancer, whether or not the cancer proved to be invasive. Only provided for train.\n\n**BIRADS** - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.\n\n**prediction_id** - The ID for the matching submission row. Multiple images will share the same prediction ID. Test only.\n\n**difficult_negative_case** - True if the case was unusually difficult. Only provided for train.","metadata":{}},{"cell_type":"code","source":"print(\"Total number of images for training are:\", len(train_df))\npatient_num = len(train_df[\"patient_id\"].unique())\nprint(\"Number of unique patients are:\", patient_num)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:03.079306Z","iopub.execute_input":"2023-03-20T13:50:03.079575Z","iopub.status.idle":"2023-03-20T13:50:03.090246Z","shell.execute_reply.started":"2023-03-20T13:50:03.07955Z","shell.execute_reply":"2023-03-20T13:50:03.089054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The total number of images are more than the number of patients. This indicates that each patient has more than 1 mammogram/image taken.","metadata":{}},{"cell_type":"code","source":"print(\"For train df\")\nmissing_values_train_info = (train_df.isnull().sum() / len(train_df)*100)\nmissing_values_train_df = pd.DataFrame()\nmissing_values_train_df['features'] = missing_values_train_info.index\nmissing_values_train_df['missing_percentage'] = missing_values_train_info.values\nmissing_values_train_df['missing_percentage'] = missing_values_train_df['missing_percentage'].apply(lambda x: f'{x:.2f}%')\nmissing_values_train_df","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:03.091704Z","iopub.execute_input":"2023-03-20T13:50:03.092545Z","iopub.status.idle":"2023-03-20T13:50:03.1195Z","shell.execute_reply.started":"2023-03-20T13:50:03.09251Z","shell.execute_reply":"2023-03-20T13:50:03.118517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"For test df\")\nmissing_values_test_info = (test_df.isnull().sum() / len(test_df)*100)\nmissing_values_test_df = pd.DataFrame()\nmissing_values_test_df['features'] = missing_values_test_info.index\nmissing_values_test_df['missing_percentage'] = missing_values_test_info.values\nmissing_values_test_df['missing_percentage'] = missing_values_test_df['missing_percentage'].apply(lambda x: f'{x:.2f}%')\nmissing_values_test_df","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:03.122026Z","iopub.execute_input":"2023-03-20T13:50:03.12265Z","iopub.status.idle":"2023-03-20T13:50:03.140791Z","shell.execute_reply.started":"2023-03-20T13:50:03.122608Z","shell.execute_reply":"2023-03-20T13:50:03.138295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"-------------------------------\")\nprint(\"Nan values in train data frame\")\nprint(\"-------------------------------\")\nprint(train_df.isna().sum())\nprint(\"-------------------------------\")\nprint(\"Nan values in test data frame\")\nprint(\"-------------------------------\")\nprint(test_df.isna().sum())","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:03.142137Z","iopub.execute_input":"2023-03-20T13:50:03.142686Z","iopub.status.idle":"2023-03-20T13:50:03.162935Z","shell.execute_reply.started":"2023-03-20T13:50:03.142648Z","shell.execute_reply":"2023-03-20T13:50:03.161361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are **no nan values for test dataframe**. \n\nIn the **train dataframe** we have **nan values for age, BIRADS and density**. \n\nSince, almost 50% data is missing for BIRADS and density therefore, I will impute the values for these columns. In case of age I will drop the nans.\n\n\n***Note:*** This imputation and dropping is just for the sake of exploratory data analysis. For model training all the data can be used because we have path to all the training images.","metadata":{}},{"cell_type":"code","source":"# drop nan in age col\ntrain_df.dropna(subset=['age'], inplace=True)\ntrain_df.reset_index(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:03.164421Z","iopub.execute_input":"2023-03-20T13:50:03.164768Z","iopub.status.idle":"2023-03-20T13:50:03.184133Z","shell.execute_reply.started":"2023-03-20T13:50:03.164734Z","shell.execute_reply":"2023-03-20T13:50:03.183221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df[\"BIRADS\"].value_counts()\n# train_df[\"density\"].value_counts()\n# convert BIRADS and density to correct data type \ntrain_df[\"BIRADS\"] = train_df['BIRADS'].astype('category')\ntrain_df[\"density\"] = train_df['density'].astype('category')\ntrain_df[[\"age\", \"BIRADS\", \"density\"]].info()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:03.185711Z","iopub.execute_input":"2023-03-20T13:50:03.186093Z","iopub.status.idle":"2023-03-20T13:50:03.219033Z","shell.execute_reply.started":"2023-03-20T13:50:03.186056Z","shell.execute_reply":"2023-03-20T13:50:03.217821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# impute values for BIRADS and density with mode value\nBIRADS_mode_value = train_df['BIRADS'].mode().iloc[0]\ndensity_mode_value = train_df['density'].mode().iloc[0]\n\ntrain_df['BIRADS'] = train_df['BIRADS'].fillna(BIRADS_mode_value)\ntrain_df['density'] = train_df['density'].fillna(density_mode_value)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:03.22463Z","iopub.execute_input":"2023-03-20T13:50:03.224927Z","iopub.status.idle":"2023-03-20T13:50:03.236699Z","shell.execute_reply.started":"2023-03-20T13:50:03.224899Z","shell.execute_reply":"2023-03-20T13:50:03.235544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot to see the count of images taken\ncount_per_patient = train_df.groupby(\"patient_id\").image_id.count()\nsns.barplot(x = count_per_patient.value_counts().index, y = count_per_patient.value_counts())\nplt.title(\"Image Count Per Patient\")","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:03.238111Z","iopub.execute_input":"2023-03-20T13:50:03.238807Z","iopub.status.idle":"2023-03-20T13:50:03.545122Z","shell.execute_reply.started":"2023-03-20T13:50:03.238771Z","shell.execute_reply":"2023-03-20T13:50:03.544055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that **most of the patients took 4-6 mammograms.**\n\nThough people who took more than 6 images are also present but the count of them is very less.","metadata":{}},{"cell_type":"code","source":"# This is total number of labels that says cancer is present\ntrain_df['cancer'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:03.548138Z","iopub.execute_input":"2023-03-20T13:50:03.548928Z","iopub.status.idle":"2023-03-20T13:50:03.558069Z","shell.execute_reply.started":"2023-03-20T13:50:03.548882Z","shell.execute_reply":"2023-03-20T13:50:03.556769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot to see how many patients got cancer\ndef has_cancer(x):\n    output = \"no_cancer(0)\"\n    if x>0:\n        output = \"cancer(1)\"\n    return output\n\ncancer_per_patient = train_df.groupby(\"patient_id\").cancer.sum().apply(lambda x: has_cancer(x))\n\nax = sns.barplot(x = cancer_per_patient.value_counts().index, y = cancer_per_patient.value_counts())\nax.bar_label(ax.containers[0])\nplt.xlabel(\"Cancer\")\nplt.ylabel(\"Count\")\nplt.title(\"Number of patients with cancer\")","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:03.559866Z","iopub.execute_input":"2023-03-20T13:50:03.56039Z","iopub.status.idle":"2023-03-20T13:50:03.793926Z","shell.execute_reply.started":"2023-03-20T13:50:03.560351Z","shell.execute_reply":"2023-03-20T13:50:03.792941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**High class imbalance**\n\nWe will perform **undersampling of majority class** (which is non_cancer represented by 0) before we go for model training.","metadata":{}},{"cell_type":"code","source":"# Plotting a histogram to see the distribution of different age groups\ndef get_val(x):\n    return x[0]\n\nage_df = train_df[['age','patient_id']]\npatient_age = age_df.groupby(\"patient_id\").age.unique().apply(lambda x: get_val(x))\nsns.histplot(patient_age.values, bins=60)\nplt.xlabel(\"Age\")\nplt.title(\"Patient age distribution\")","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:03.796094Z","iopub.execute_input":"2023-03-20T13:50:03.797181Z","iopub.status.idle":"2023-03-20T13:50:04.652128Z","shell.execute_reply.started":"2023-03-20T13:50:03.797139Z","shell.execute_reply":"2023-03-20T13:50:04.651078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the females getting a mammogram done are in the age range 40-70.\n\nFurther, we can explore the distribution of females having cancer with those not having cancer in the various age groups using the plot below.","metadata":{}},{"cell_type":"code","source":"age_cancer = train_df.groupby([\"patient_id\",\"cancer\"]).age.unique().apply(lambda x:get_val(x))\nage_cancer_df = pd.DataFrame(age_cancer).reset_index()\n\nfig,ax = plt.subplots(figsize=(10,5))\nsns.histplot(age_cancer_df[age_cancer_df[\"cancer\"] == 0][\"age\"], color=\"b\",bins=60,kde=True,ax=ax)\nax.set_ylabel(\"No cancer count(blue)\")\n\nax2 = ax.twinx()\nax2.set_ylabel(\"Cancer count(red)\")\nsns.histplot(age_cancer_df[age_cancer_df[\"cancer\"] == 1][\"age\"], color=\"r\", bins=60, kde=True, ax=ax2)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:04.653822Z","iopub.execute_input":"2023-03-20T13:50:04.654211Z","iopub.status.idle":"2023-03-20T13:50:06.380223Z","shell.execute_reply.started":"2023-03-20T13:50:04.654172Z","shell.execute_reply":"2023-03-20T13:50:06.379069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the count of females with implants having cancer\npatient_implant = train_df.groupby(\"patient_id\").implant.max()\nax = sns.barplot(x = patient_implant.value_counts().index, y = patient_implant.value_counts())\nax.bar_label(ax.containers[0])\nplt.title(\"Patients with and without implant\")\nax.set_xticklabels([\"No_implant(0)\", \"Implant(1)\"])\n\npatient_with_implant = len(patient_implant[patient_implant.values>0])\nprint(f\"There are {patient_with_implant} ({round(100*patient_with_implant/patient_num, 1)}%) patients having implant.\")","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:06.381875Z","iopub.execute_input":"2023-03-20T13:50:06.38226Z","iopub.status.idle":"2023-03-20T13:50:06.607651Z","shell.execute_reply.started":"2023-03-20T13:50:06.382222Z","shell.execute_reply":"2023-03-20T13:50:06.606603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot to see which machine was most used for diagnosing cancer. \nmachine_df = train_df.groupby(\"machine_id\").count()[\"patient_id\"].reset_index()\nplt.figure(figsize=(10,5))\nsns.barplot(x=\"machine_id\", y=\"patient_id\", data=machine_df)\nplt.ylabel(\"Count\")\nplt.title(\"Which machine was most used for diagnosing cancer...?\")","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:06.609343Z","iopub.execute_input":"2023-03-20T13:50:06.609744Z","iopub.status.idle":"2023-03-20T13:50:06.889984Z","shell.execute_reply.started":"2023-03-20T13:50:06.609704Z","shell.execute_reply":"2023-03-20T13:50:06.889044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Machine 49** was **used the most** for diagnosing cancer. \n\nMachine 21, 29, and 48 were used almost equally.\n\nThe following plot illustrates (in percentage) the cancer cases found on the above machines.","metadata":{}},{"cell_type":"code","source":"machine_cancer_df = train_df.groupby([\"machine_id\", \"cancer\"]).count()[\"patient_id\"].reset_index()\ntotal_machine_df = machine_cancer_df.groupby(\"machine_id\").sum().reset_index()\nmachine_cancer_percent = machine_cancer_df.merge(total_machine_df, on=\"machine_id\")\nmachine_cancer_percent[\"cancer_percent\"] = round(100.0*machine_cancer_percent[\"patient_id_x\"]/machine_cancer_percent[\"patient_id_y\"],1)\ncancer_percent = machine_cancer_percent[machine_cancer_percent[\"cancer_x\"]==1]\nplt.figure(figsize=(10,5))\nax = sns.barplot(x=\"machine_id\", y=\"cancer_percent\", data=cancer_percent)\nax.bar_label(ax.containers[0])\nplt.ylabel(\"Cancer percent(%)\")\nplt.title(\"Percentage of cancer detected on the machines\")","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:06.891353Z","iopub.execute_input":"2023-03-20T13:50:06.891706Z","iopub.status.idle":"2023-03-20T13:50:07.218953Z","shell.execute_reply.started":"2023-03-20T13:50:06.891667Z","shell.execute_reply":"2023-03-20T13:50:07.217952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cancer detected on machine 21, 29, 48, 49 is around **2% ~ 2.5%** which is quite close. ","metadata":{}},{"cell_type":"markdown","source":"One might get mistaken by the thing that machine 190 detects highest cancer. In real, machine 190 had 1 or very few patients, therefore the cancer detection seems high.","metadata":{"execution":{"iopub.status.busy":"2023-03-12T04:04:18.810989Z","iopub.execute_input":"2023-03-12T04:04:18.811399Z","iopub.status.idle":"2023-03-12T04:04:18.818833Z","shell.execute_reply.started":"2023-03-12T04:04:18.811359Z","shell.execute_reply":"2023-03-12T04:04:18.81745Z"}}},{"cell_type":"code","source":"# Plotting images with and without cancer. Reference https://www.kaggle.com/code/dingyan/rsna-eda-pca-logistic-regression\n\nimage_no_cancer = train_df[train_df[\"cancer\"]==0]\nimage_cancer = train_df[train_df[\"cancer\"]==1]\n\n# Paths to image data\ntrain_images_path = \"/kaggle/input/rsna-breast-cancer-detection/train_images/\"\ntest_images_path = \"/kaggle/input/rsna-breast-cancer-detection/test_images/\"\n\ndef get_image_path(image_df, n):\n    ids = image_df[[\"patient_id\", \"image_id\"]].sample(n, random_state=42)\n    paths = []\n    for i in range(len(ids)):\n        path = os.path.join(train_images_path, str(ids[\"patient_id\"].values[i]), str(ids[\"image_id\"].values[i])+'.dcm')\n        paths.append(path)\n    return paths\n\nn = 5\nno_cancer_paths = get_image_path(image_no_cancer, n)\ncancer_paths = get_image_path(image_cancer, n)\n\nplt.figure(figsize=(15, 8))\n\nfor j in range(n):\n    plt.subplot(1, n, j+1)\n    im = dicomsdl.open(no_cancer_paths[j])\n    plt.imshow(im.pixelData(storedvalue=False), cmap='bone') \n    plt.grid(False)\n    plt.title(f\"No cancer\")\n    plt.axis(\"off\")\nplt.show()\nplt.figure(figsize=(15, 8))\n\nfor j in range(n):\n    plt.subplot(1, n, j + 1)\n    im = dicomsdl.open(cancer_paths[j])\n    plt.imshow(im.pixelData(storedvalue=False), cmap='bone')\n    plt.grid(False)\n    plt.title(f\"Cancer\")\n    plt.axis(\"off\")\nplt.show()\nfig.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:07.220972Z","iopub.execute_input":"2023-03-20T13:50:07.221527Z","iopub.status.idle":"2023-03-20T13:50:19.80243Z","shell.execute_reply.started":"2023-03-20T13:50:07.221488Z","shell.execute_reply":"2023-03-20T13:50:19.800863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define functions and classes - preprocessing, model","metadata":{}},{"cell_type":"markdown","source":"## Important points\n\n1. Offset correction is important to do so that the range of pixel values in the image starts from zero. This is done by subtracting the minimum pixel value of the image from every pixel.\n2. After offset correction we normalize our pixel values so that they lie in the range of 0 to 1. This makes it easier to work with subsequent steps.\n3. Some of the images can be inverted. If the X-ray image is in \"MONOCHROME1\" format, then it needs to be inverted so that the darkest pixels become the lightest and vice versa. This is done by subtracting the maximum pixel value in the image from every pixel.\n4. We perform min max scaling so as to ensure that after the inversion correction, the pixel values are still in the range of 0 to 1.\n5. We then convert the pre-processed image to 8 bit grayscale image to reduce the memory requirements of the image. This make it easier to work with the subsequent steps.\n6. Honestly, I am not sure how people arrived at a value of 5 for image cropping. But, here is the reference where I first saw this https://www.kaggle.com/code/atshimamura/jpn-getting-started-with-keras-low-score/notebook\n","metadata":{}},{"cell_type":"code","source":"\"\"\"\nThe function below reads a dicom file, performs the offset correction, fix any inverted images and\nfinally performs min max scaling. \n\"\"\"\n\ndef normalize_xray(path, fix_monochrome = True):\n    dicom = dicomsdl.open(path)\n    dicom_arr = dicom.pixelData(storedvalue=False)  # storedvalue = True for int16 return otherwise float32    \n    #offset correction\n    dicom_arr = dicom_arr - np.min(dicom_arr)    \n    dicom_arr = dicom_arr / np.max(dicom_arr)    \n    # fix the inversion\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        dicom_arr = np.amax(dicom_arr) - dicom_arr    \n    # Min Max Scaling\n    dicom_arr = dicom_arr - np.min(dicom_arr)\n    dicom_arr = dicom_arr / np.max(dicom_arr)\n    dicom_arr = (dicom_arr * 255).astype(np.uint8) # convert to 8-bit grayscale image\n    return dicom_arr\n\n\"\"\"\nSome images have unnecessary borders, so we remove them using the\nfunction below.\n\"\"\"\ndef crop_and_resize(image, crop_size=5):\n    image = image[crop_size:-crop_size, crop_size:-crop_size]\n    image = cv2.resize(image, target_size[::-1], cv2.INTER_LINEAR)\n    return image\n\n\"\"\"\nThis function calls the normalization function and crop image function defined above.\n\"\"\"\ndef preprocess_images(file_path):\n    image = normalize_xray(file_path)\n    h, w = image.shape[:2]  # take the height and width\n    image = crop_and_resize(image)\n    sub_path = file_path.split(\"/\",4)[-1].split('.dcm')[0] + '.png'\n    infos = sub_path.split('/')\n    pid = infos[-2]\n    iid = infos[-1]; iid = iid.replace('.png','')\n    return pid, iid, h, w, image\n\n\"\"\"\nThe function below defines our competition evaluation metric - probablistic f1 score\nhttps://www.kaggle.com/code/awsaf49/metric-probabilistic-fscore-tf-torch-numpy\n\"\"\"\ndef pfbeta_torch(labels, preds, beta=1):\n    preds = preds.clip(0, 1)\n    y_true_count = labels.sum()\n    ctp = preds[labels==1].sum()\n    cfp = preds[labels==0].sum()\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0.0","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:19.808925Z","iopub.execute_input":"2023-03-20T13:50:19.810059Z","iopub.status.idle":"2023-03-20T13:50:19.841774Z","shell.execute_reply.started":"2023-03-20T13:50:19.809979Z","shell.execute_reply":"2023-03-20T13:50:19.8401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nThe code below defines a custom PyTorch dataset called \"MyDataset\" for training/validation purposes.\n\nThe __init__ method initializes the dataset with a Pandas DataFrame containing the path to DICOM files and\ntheir corresponding cancer labels. It also takes an optional transform argument that can be used to apply \ntransformations to the images.\n\nThe __len__ method returns the length of the dataset, which is the number of samples in the DataFrame.\n\nThe __getitem__ method returns a single sample from the dataset at the given index.\nIt first retrieves the cancer label and DICOM file path from the DataFrame.\nIt then preprocesses the DICOM file by extracting the patient ID, image ID, height, width, and pixel array\nusing the preprocess_images function. The pixel array is then converted into a PIL Image object and transformed\n(if a transform was specified) before being returned as a dictionary containing the cancer label and the transformed image tensor.\n\"\"\"\n\nclass MyDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        cancer = row['cancer']\n        cancer = torch.tensor(cancer, dtype=torch.long) \n        \n        path = row['dcm_path']\n        path_parts = path.split('/')\n        filename = path_parts[-1]\n        full_path = os.path.join('/'.join(path_parts[:-1]), filename)\n        \n        # Read the DICOM file and preprocess it\n        pid, iid, h, w, img = preprocess_images(full_path)\n        \n        img = Image.fromarray(img).convert('RGB')\n        if self.transform:\n            img = self.transform(img)\n        return {'cancer': cancer, 'images': img} \n    \n\"\"\"\nThis is a class for creating a PyTorch dataset for testing purpose. The class takes a DataFrame \nas an input, which contains information about the test images. The __init__ function initializes the dataset\nby storing the DataFrame and any transform to be applied to the images during training or inference.\n\nThe __len__ function returns the number of items in the dataset.\n\nThe __getitem__ function is called when an item is accessed by index. It retrieves the corresponding row from\nthe DataFrame, extracts the path to the image, reads and preprocesses the image, applies any transforms, and\nreturns the patient ID and image.\n\n\"\"\"\n\nclass TestDataset(Dataset):\n    def __init__(self, df, transform=None):\n        self.df = df\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        path = row['dcm_path']\n        path_parts = path.split('/')\n        patient_id = path_parts[-2]\n        filename = path_parts[-1]\n        full_path = os.path.join('/'.join(path_parts[:-1]), filename)\n        # Read the DICOM file and preprocess \n        pid, iid, h, w, img = preprocess_images(full_path)\n        img = Image.fromarray(img).convert('RGB')\n        if self.transform:\n            img = self.transform(img)\n        return patient_id, img","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:19.843699Z","iopub.execute_input":"2023-03-20T13:50:19.844358Z","iopub.status.idle":"2023-03-20T13:50:19.86868Z","shell.execute_reply.started":"2023-03-20T13:50:19.84432Z","shell.execute_reply":"2023-03-20T13:50:19.867044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Helpful points\n\nThis is a PyTorch model defined as a subclass of nn.Module. The model is called PretrainedBinaryClassifier and it uses a pre-trained EfficientNet model for binary classification.\nThe efficientnet-b0 architecture used in this model has a total of 20 layers, including a stem convolutional layer, a sequence of 5 blocks of convolutional layers with increasing depth and width, and a global average pooling layer. The number of channels in each block increases while the resolution decreases, resulting in feature maps with progressively larger receptive fields. The final output of the model is a linear layer that produces a single scalar value, which is used as the input to the sigmoid activation function for binary classification.\n\n1. super(PretrainedBinaryClassifier, self).__init__(): This line calls the constructor of the parent class nn.Module to properly initialize the model.\n\n2. self.model = EfficientNet.from_pretrained('efficientnet-b0', num_classes=1, weights_path=\"/kaggle/input/model-files/efficientnet-b0-355c32eb.pth\"): This line loads the pre-trained EfficientNet model with the architecture efficientnet-b0 from the torchvision.models library, and sets the number of output classes to 1 for binary classification. The weights_path argument specifies the path to the pre-trained weights file.\n\n3. def forward(self, x):: This function defines the forward pass of the model. It takes an input tensor x and returns the output tensor.\n\n4. x = self.model(x): This line passes the input tensor x through the pre-trained EfficientNet model, producing an output tensor.\n\n5. x = torch.sigmoid(x): This line applies a sigmoid function to the output tensor to squash its values between 0 and 1, which is appropriate for binary classification.\n\n6. return x: This line returns the output tensor, which contains the binary classification scores for each input image.\n\nFor the data augmentation part, I tried some horizontal flips as follows\n\ntransforms.Compose([\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomRotation(10),\n    transforms.ToTensor()\n])\n\nBut, in hope of getting better results I copied the augmentation of 3rd place solution (https://www.kaggle.com/competitions/rsna-breast-cancer-detection/discussion/391779)","metadata":{}},{"cell_type":"code","source":"\"\"\"\nDefining my model class.\nFor this competition I am using EfficientNet pre-trained model\n\"\"\"\n\nclass PretrainedBinaryClassifier(nn.Module):\n    def __init__(self):\n        super(PretrainedBinaryClassifier, self).__init__()\n        # Load the pre-trained EfficientNet model\n        self.model = EfficientNet.from_pretrained('efficientnet-b0', num_classes=1, weights_path=\"/kaggle/input/model-files/efficientnet-b0-355c32eb.pth\")\n        \n    def forward(self, x):\n        x = self.model(x)\n        x = torch.sigmoid(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:19.870595Z","iopub.execute_input":"2023-03-20T13:50:19.871225Z","iopub.status.idle":"2023-03-20T13:50:19.883474Z","shell.execute_reply.started":"2023-03-20T13:50:19.871187Z","shell.execute_reply":"2023-03-20T13:50:19.881959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nOne of the transformations (transforms.RandomErasing) may sometimes remove parts of the image, \nwhich could result in an image with one or more dimensions being zero. When this happens, the transforms.ToTensor() operation fails,\nbecause it expects the image to have at least one non-zero dimension.\n\nTo solve this we first convert the PIL image to a tensor and then apply the transforms.RandomErasing() before normalizing. \nWith this approach, the transforms.RandomErasing() is applied to the tensor representation of the image rather than the PIL image, \nso it won't cause issues with the transforms.ToTensor() operation.\n\"\"\"\nclass ToTensorWithErasing(object):\n    def __call__(self, img):\n        img_tensor = transforms.functional.to_tensor(img)\n        img_tensor = transforms.RandomErasing(p=0.5, scale=(0.1, 0.5), ratio=(0.3, 3.3))(img_tensor)\n        return img_tensor\n\n\"\"\"\nDefining our transformations for train, validation and test sets. \nWe don't perform data augmentation on validation sets.\n\nhttps://www.kaggle.com/competitions/rsna-breast-cancer-detection/discussion/391779\n\"\"\"\n\ntrain_transform = transforms.Compose([\n    transforms.RandomHorizontalFlip(p=0.5),\n    transforms.RandomVerticalFlip(p=0.5),\n    transforms.ColorJitter(brightness=0.1, contrast=0.1, saturation=0.1, hue=0.1),\n    transforms.RandomAffine(degrees=45, translate=(0.3, 0.3), scale=(0.7, 1.3)),\n    transforms.RandomPerspective(distortion_scale=0.3, p=0.7),\n    ToTensorWithErasing(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\n\nval_transform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])\n])\n\ntest_transform = transforms.Compose([transforms.ToTensor()])","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:19.8852Z","iopub.execute_input":"2023-03-20T13:50:19.885994Z","iopub.status.idle":"2023-03-20T13:50:19.903105Z","shell.execute_reply.started":"2023-03-20T13:50:19.885953Z","shell.execute_reply":"2023-03-20T13:50:19.90143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create training and validation sets","metadata":{}},{"cell_type":"markdown","source":"## Undersampling","metadata":{}},{"cell_type":"code","source":"# set seed for reproducibility\nrandom_seed = 534\ntrain_subset_0 = train_df[train_df.cancer == 0].sample(n = 460, random_state=random_seed) #460\ntrain_subset_1 = train_df[train_df.cancer == 1].sample(n = 360, random_state=random_seed) # 360\ntrain_combined = pd.concat([train_subset_0, train_subset_1])\n\nprint(train_combined.shape)\nprint(train_combined.cancer.value_counts())\ntrain_combined.reset_index(inplace=True)\n\n# train test split\ntraining_set, validation_set = train_test_split(train_combined, test_size=0.2, random_state=276)\nprint(training_set.shape)\nprint(validation_set.shape)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:19.904908Z","iopub.execute_input":"2023-03-20T13:50:19.90573Z","iopub.status.idle":"2023-03-20T13:50:19.94286Z","shell.execute_reply.started":"2023-03-20T13:50:19.905693Z","shell.execute_reply":"2023-03-20T13:50:19.941813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create pytorch training and validation datasets","metadata":{}},{"cell_type":"code","source":"%%time\n\ntrain_dataset = MyDataset(training_set, transform = train_transform) \nval_dataset = MyDataset(validation_set, transform = val_transform) ","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:19.943982Z","iopub.execute_input":"2023-03-20T13:50:19.944337Z","iopub.status.idle":"2023-03-20T13:50:19.953446Z","shell.execute_reply.started":"2023-03-20T13:50:19.944298Z","shell.execute_reply":"2023-03-20T13:50:19.952204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Checking if pytorch datasets get created...\")\nif len(train_dataset) > 0:\n    print(\"The training dataset contains\", len(train_dataset), \"samples.\")\nelse:\n    print(\"The training dataset is empty.\")\n    \nif len(val_dataset) > 0:\n    print(\"The val dataset contains\", len(val_dataset), \"samples.\")\nelse:\n    print(\"The val dataset is empty.\")","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:19.955165Z","iopub.execute_input":"2023-03-20T13:50:19.955524Z","iopub.status.idle":"2023-03-20T13:50:19.96398Z","shell.execute_reply.started":"2023-03-20T13:50:19.955489Z","shell.execute_reply":"2023-03-20T13:50:19.962945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Checking if transformation is applied\")\nprint(\"Training dataset transform is: \", train_dataset.transform)\nprint(\"Validation dataset transform is:\", val_dataset.transform)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:19.965478Z","iopub.execute_input":"2023-03-20T13:50:19.965887Z","iopub.status.idle":"2023-03-20T13:50:19.975163Z","shell.execute_reply.started":"2023-03-20T13:50:19.965851Z","shell.execute_reply":"2023-03-20T13:50:19.973519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create train and val data loaders","metadata":{}},{"cell_type":"code","source":"%%time\n\n\"\"\"\nnum_workers=1 implies that one additional worker process will be created to\nload the data in parallel with the main process.\n\"\"\"\n\ntrain_loader = DataLoader(train_dataset, batch_size = batch_size, shuffle = True, num_workers = 1)\nval_loader = DataLoader(val_dataset, batch_size = batch_size, shuffle = False)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:19.977028Z","iopub.execute_input":"2023-03-20T13:50:19.977468Z","iopub.status.idle":"2023-03-20T13:50:19.986028Z","shell.execute_reply.started":"2023-03-20T13:50:19.977428Z","shell.execute_reply":"2023-03-20T13:50:19.984735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nNumber of batches = len(train_dataset)/batch_size. For e.g.: - 264/64 = 4.125 which means\n4 full batches and 0.125 extra batch in order to read all the 264 training samples.\nSo these 4.125 batches are passed through each epoch. The weights are updated after each epoch and\nthe model continues to learn.\n\"\"\"\n\nprint(\"Number of training batches\",len(train_loader)) \nprint(\"Number of validation batches\",len(val_loader))","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:19.987886Z","iopub.execute_input":"2023-03-20T13:50:19.988978Z","iopub.status.idle":"2023-03-20T13:50:19.995199Z","shell.execute_reply.started":"2023-03-20T13:50:19.98894Z","shell.execute_reply":"2023-03-20T13:50:19.994068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create an instance of model, train and evaluate it","metadata":{}},{"cell_type":"code","source":"# Checking if gpu is available\nprint(torch.cuda.is_available())\n# print the number of available GPUs on the current system. If no GPUs are available, it would print 0.\nprint(torch.cuda.device_count()) ","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:19.996544Z","iopub.execute_input":"2023-03-20T13:50:19.997584Z","iopub.status.idle":"2023-03-20T13:50:20.131928Z","shell.execute_reply.started":"2023-03-20T13:50:19.997542Z","shell.execute_reply":"2023-03-20T13:50:20.130764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Useful points\n\nnn.BCEWithLogitsLoss() is a PyTorch loss function that is commonly used for binary classification problems. It combines the binary cross-entropy (BCE) loss and the sigmoid activation function into a single function that is more numerically stable and efficient to compute.\n","metadata":{}},{"cell_type":"code","source":"device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n\nmodel = PretrainedBinaryClassifier()\n# Move the model to the GPU device if it is available\nmodel.to(device)\n\n# Define loss function https://pytorch.org/docs/stable/generated/torch.nn.BCEWithLogitsLoss.html\ncriterion = nn.BCEWithLogitsLoss() \ncriterion = criterion.cuda() # move to GPU \n\nlearning_rate = 0.0001\nweight_decay = 0.01\n\n# Define optimizer\noptimizer = optim.Adam(model.parameters(), lr=learning_rate, weight_decay=weight_decay)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:20.133591Z","iopub.execute_input":"2023-03-20T13:50:20.133951Z","iopub.status.idle":"2023-03-20T13:50:25.105723Z","shell.execute_reply.started":"2023-03-20T13:50:20.133914Z","shell.execute_reply":"2023-03-20T13:50:25.104683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ! pip install torchsummary\n# from torchsummary import summary\n# # Display the model summary\n# summary(model, input_size=(3, 224, 224))","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:25.107158Z","iopub.execute_input":"2023-03-20T13:50:25.108203Z","iopub.status.idle":"2023-03-20T13:50:25.112913Z","shell.execute_reply.started":"2023-03-20T13:50:25.108162Z","shell.execute_reply":"2023-03-20T13:50:25.11156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Train the model for multiple epochs\ntrain_losses = []\ntrain_acc_metric = []\ntrain_pf1_metric = []\n\nval_losses = []\nval_acc_metric = []\nval_pf1_metric = []\n\nfor epoch in range(num_epochs):\n    # Train\n    running_train_loss = 0.0\n    running_train_acc = 0.0\n    running_train_pf1 = 0\n    \n    model.train() # set the model to training mode\n    for i, data in enumerate(train_loader, 0):\n        inputs, targets = data['images'], data['cancer']\n        targets = targets.view(-1, 1) # Reshape the target tensor to (batch_size, 1)\n        inputs = inputs.to(device)\n        targets = targets.to(device)\n        \n        # Zero the parameter gradients\n        optimizer.zero_grad()\n        # Forward pass\n        outputs = model(inputs)\n        # Compute loss\n        loss = criterion(outputs, targets.float())\n        \n        # Compute training accuracy\n        predicted = torch.round(torch.sigmoid(outputs))\n        correct = (predicted == targets).sum().item()\n        accuracy = correct / targets.size(0)\n        # Compute pf1 metric\n        pf1_metric = pfbeta_torch(targets.cpu().numpy(), torch.sigmoid(outputs).detach().cpu().numpy())\n\n        # Backward pass and optimize\n        loss.backward()\n        optimizer.step()\n\n        # Accumulate loss and metric for the epoch\n        running_train_loss += loss.item()\n        running_train_acc += accuracy\n        running_train_pf1 += pf1_metric\n\n    # Compute and print average loss and metrics for the epoch\n    avg_train_loss = running_train_loss / len(train_loader)\n    avg_train_acc = running_train_acc / len(train_loader)\n    avg_train_pf1 = running_train_pf1 / len(train_loader)\n    train_losses.append(avg_train_loss)\n    train_acc_metric.append(avg_train_acc) # train_metrics\n    train_pf1_metric.append(avg_train_pf1)\n    print('Epoch %d, avg training loss: %.3f, avg training accuracy: %.3f, avg training pf1: %.3f'%\n          (epoch + 1, avg_train_loss, avg_train_acc, avg_train_pf1))\n\n\n    # Validate\n    model.eval() # set the model to evaluation mode\n    running_val_loss = 0.0\n    running_val_acc = 0.0\n    running_val_pf1 = 0\n\n    with torch.no_grad():\n        for i, data in enumerate(val_loader, 0):\n            inputs, targets = data['images'], data['cancer']\n            targets = targets.view(-1, 1) # Reshape the target tensor to (batch_size, 1)\n            inputs = inputs.to(device)\n            targets = targets.to(device)\n\n            # Forward pass\n            outputs = model(inputs)\n\n            # Compute loss\n            loss = criterion(outputs, targets.float())\n\n            # Compute validation accuracy\n            predicted = torch.round(torch.sigmoid(outputs))\n            correct = (predicted == targets).sum().item()\n            accuracy = correct / targets.size(0)\n            \n            # Compute pf1 metric\n            pf1_metric = pfbeta_torch(targets.cpu().numpy(), torch.sigmoid(outputs).detach().cpu().numpy())\n\n            # Accumulate loss and metric for the epoch\n            running_val_loss += loss.item()\n            running_val_acc += accuracy\n            running_val_pf1 += pf1_metric\n\n    # Compute and print average loss and accuracy for validation set\n    avg_val_loss = running_val_loss / len(val_loader)\n    avg_val_acc = running_val_acc / len(val_loader)\n    avg_val_pf1 = running_val_pf1 / len(val_loader)\n    val_losses.append(avg_val_loss)\n    val_acc_metric.append(avg_val_acc)\n    val_pf1_metric.append(avg_val_pf1)\n    print('Epoch %d, avg validation loss: %.3f, avg validation accuracy: %.3f, avg validation pf1: %.3f' %\n          (epoch + 1, avg_val_loss, avg_val_acc, avg_val_pf1))","metadata":{"execution":{"iopub.status.busy":"2023-03-20T13:50:25.114622Z","iopub.execute_input":"2023-03-20T13:50:25.115048Z","iopub.status.idle":"2023-03-20T14:46:40.222454Z","shell.execute_reply.started":"2023-03-20T13:50:25.114991Z","shell.execute_reply":"2023-03-20T14:46:40.22104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the learning curves \n\nplt.plot(train_losses, label='Training Loss')\nplt.plot(val_losses, label='Validation Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\n\nplt.plot(train_acc_metric, label='Training Accuracy')\nplt.plot(val_acc_metric, label='Validation Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\n\nplt.plot(train_pf1_metric, label='Training Probabilistic F1 Score')\nplt.plot(val_pf1_metric, label='Validation Probabilistic F1 Score')\nplt.xlabel('Epoch')\nplt.ylabel('Probabilistic F1 Score')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:46:40.224591Z","iopub.execute_input":"2023-03-20T14:46:40.226292Z","iopub.status.idle":"2023-03-20T14:46:40.900994Z","shell.execute_reply.started":"2023-03-20T14:46:40.22624Z","shell.execute_reply":"2023-03-20T14:46:40.899947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction on test set","metadata":{}},{"cell_type":"code","source":"%%time\n\n\"\"\"\nJust like training data, create pytorch dataset and data loader for\ntest data. \n\"\"\"\n\ntest_dataset = TestDataset(test_df, transform = test_transform)\ntest_dataloader = DataLoader(test_dataset, batch_size = test_batch_size, shuffle=False, num_workers = 1)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:46:40.902625Z","iopub.execute_input":"2023-03-20T14:46:40.902984Z","iopub.status.idle":"2023-03-20T14:46:40.909734Z","shell.execute_reply.started":"2023-03-20T14:46:40.902946Z","shell.execute_reply":"2023-03-20T14:46:40.908467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Initialize an empty list to store predictions\npredictions = []\n\n# Iterate through the test dataloader\nfor batch in test_dataloader:\n    pids, images = batch\n    images = images.to(device)\n    \n    # Forward pass\n    with torch.no_grad():\n        outputs = model(images)\n        probabilities = torch.sigmoid(outputs)\n        predictions.extend(probabilities.cpu().numpy())","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:46:40.911573Z","iopub.execute_input":"2023-03-20T14:46:40.911922Z","iopub.status.idle":"2023-03-20T14:46:43.781092Z","shell.execute_reply.started":"2023-03-20T14:46:40.911886Z","shell.execute_reply":"2023-03-20T14:46:43.778757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save predictions to dataframe and create submission csv file","metadata":{}},{"cell_type":"code","source":"def get_val(x):\n    return x[0]\n\ntest_df['cancer'] = predictions\ntest_df['cancer'] = test_df['cancer'].apply(lambda x: get_val(x))\ntest_df.tail()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:46:43.792792Z","iopub.execute_input":"2023-03-20T14:46:43.796672Z","iopub.status.idle":"2023-03-20T14:46:43.818998Z","shell.execute_reply.started":"2023-03-20T14:46:43.796616Z","shell.execute_reply":"2023-03-20T14:46:43.817837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = test_df.groupby('prediction_id')['cancer'].mean().to_frame().reset_index()\nsub.tail()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:46:43.821032Z","iopub.execute_input":"2023-03-20T14:46:43.821978Z","iopub.status.idle":"2023-03-20T14:46:43.841095Z","shell.execute_reply.started":"2023-03-20T14:46:43.821935Z","shell.execute_reply":"2023-03-20T14:46:43.840075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:46:43.842532Z","iopub.execute_input":"2023-03-20T14:46:43.842983Z","iopub.status.idle":"2023-03-20T14:46:43.855899Z","shell.execute_reply.started":"2023-03-20T14:46:43.842943Z","shell.execute_reply":"2023-03-20T14:46:43.854849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Useful Tests","metadata":{}},{"cell_type":"code","source":"# \"\"\"\n# Code to check if the augmentations get applied correcty or not.\n\n# Note: The images present here in the train_loader should be unaugmented one. \n# \"\"\"\n\n# for i, data in enumerate(train_loader, 0):\n#     inputs, targets = data['images'], data['cancer']\n\n#     # Convert tensor inputs to PIL Images\n#     inputs_pil = [ToPILImage()(img) for img in inputs]\n\n#     # Plot the original images\n#     fig, axs = plt.subplots(1, 3, figsize=(10, 5))\n#     for i in range(3):\n#         axs[i].imshow(inputs_pil[i])\n#         axs[i].axis('off')\n#         axs[i].set_title('Original')\n    \n#     # Apply the transform to the images\n#     images_aug = [train_transform(image) for image in inputs_pil]\n\n#     # Plot the augmented images\n#     fig, axs = plt.subplots(1, 3, figsize=(10, 5))\n#     for i in range(3):\n#         #axs[i].imshow(images_aug[i])\n#         axs[i].imshow(images_aug[i].permute(1, 2, 0))\n#         axs[i].axis('off')\n#         axs[i].set_title('Augmented')\n","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:46:43.858923Z","iopub.execute_input":"2023-03-20T14:46:43.859357Z","iopub.status.idle":"2023-03-20T14:46:43.864915Z","shell.execute_reply.started":"2023-03-20T14:46:43.859318Z","shell.execute_reply":"2023-03-20T14:46:43.863645Z"},"trusted":true},"execution_count":null,"outputs":[]}]}