{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"You can ignore most of the imports for now. I copied them from the example notebook. They will be used mostly down the road","metadata":{}},{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:46:16.80492Z","iopub.execute_input":"2022-11-30T06:46:16.805368Z","iopub.status.idle":"2022-11-30T06:46:36.150133Z","shell.execute_reply.started":"2022-11-30T06:46:16.805331Z","shell.execute_reply":"2022-11-30T06:46:36.148833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing the training images and previewing them","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nsns.set()\nfrom PIL import Image\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau, StepLR, CyclicLR\nimport torchvision\nfrom torchvision import datasets, models, transforms\nfrom torch.utils.data import Dataset, DataLoader\nimport torch.nn.functional as F\n\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.utils.class_weight import compute_class_weight\n\n\nfrom glob import glob\nfrom skimage.io import imread\nfrom os import listdir\n\nimport time\nimport copy\nfrom tqdm import tqdm_notebook as tqdm\n\nimport pydicom # used for viewing the .dcm files given\nimport gdcm\nimport pylibjpeg\nfrom joblib import Parallel, delayed","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:46:36.153514Z","iopub.execute_input":"2022-11-30T06:46:36.154045Z","iopub.status.idle":"2022-11-30T06:46:39.585517Z","shell.execute_reply.started":"2022-11-30T06:46:36.153984Z","shell.execute_reply":"2022-11-30T06:46:39.584618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Access to the directory for the main files given by the kaggle\nbase_path = \"/kaggle/input/rsna-breast-cancer-detection/\"\nfiles = listdir(base_path)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:46:39.587028Z","iopub.execute_input":"2022-11-30T06:46:39.588359Z","iopub.status.idle":"2022-11-30T06:46:39.594618Z","shell.execute_reply.started":"2022-11-30T06:46:39.588319Z","shell.execute_reply":"2022-11-30T06:46:39.592902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defines the path to each data file or directory \ntest_images_path = base_path + \"test_images\"\ntest_files = listdir(test_images_path)\n\ntrain_images_path = base_path + \"train_images\"\ntrain_images_files = listdir(train_images_path)\n\n# Data paths\ntrain_data_path = base_path + \"/train.csv\"\ntest_data_path = base_path + \"/test.csv\"\nsample_submission_path = base_path + \"/sample_submission.csv\"","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:46:39.597978Z","iopub.execute_input":"2022-11-30T06:46:39.598713Z","iopub.status.idle":"2022-11-30T06:46:39.763301Z","shell.execute_reply.started":"2022-11-30T06:46:39.598658Z","shell.execute_reply":"2022-11-30T06:46:39.762279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dataframe for the training data\npatient_data = pd.read_csv(train_data_path)\npatient_data","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:46:39.764637Z","iopub.execute_input":"2022-11-30T06:46:39.765151Z","iopub.status.idle":"2022-11-30T06:46:39.913934Z","shell.execute_reply.started":"2022-11-30T06:46:39.765118Z","shell.execute_reply":"2022-11-30T06:46:39.91271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adds the filepath to the dataframe using the patient_id \n# Returns the dataframe with the file_path column added\ndef add_filepath(df):\n    # Creates the column\n    df[\"file_path\"] = train_images_path + \"/\" + patient_data[\"patient_id\"].astype(str) + \"/\" + patient_data[\"image_id\"].astype(str) + \".dcm\"\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:46:39.916376Z","iopub.execute_input":"2022-11-30T06:46:39.9169Z","iopub.status.idle":"2022-11-30T06:46:39.922808Z","shell.execute_reply.started":"2022-11-30T06:46:39.916866Z","shell.execute_reply":"2022-11-30T06:46:39.921396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the patient_df\npatient_df = add_filepath(patient_data)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:46:39.924086Z","iopub.execute_input":"2022-11-30T06:46:39.924429Z","iopub.status.idle":"2022-11-30T06:46:40.0494Z","shell.execute_reply.started":"2022-11-30T06:46:39.9244Z","shell.execute_reply":"2022-11-30T06:46:40.048132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reads the image in the specified file paths\n# Plots an output image|\n# Credit to Theo Viel\n\nfor f in tqdm(patient_data[\"file_path\"][190:191]):\n    patient = f.split('/')[-2]\n    image = f.split('/')[-1][:-4]\n\n    dicom = pydicom.dcmread(f)\n    img = dicom.pixel_array\n\n    img = (img - img.min()) / (img.max() - img.min())\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n        \n    plt.figure(figsize=(15, 15))\n    plt.imshow(img, cmap=\"gray\")\n    plt.title(f\"{patient} {image}\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T07:14:01.214699Z","iopub.execute_input":"2022-11-30T07:14:01.215169Z","iopub.status.idle":"2022-11-30T07:14:03.657036Z","shell.execute_reply.started":"2022-11-30T07:14:01.21513Z","shell.execute_reply":"2022-11-30T07:14:03.655702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring the Data","metadata":{}},{"cell_type":"markdown","source":"### Using the patients_df explore characteristics of the data\n* Is the data clean?\n* Is the data balanced?\n* Is the data normal? \n* What is the attribute selection? - correlation","metadata":{"execution":{"iopub.status.busy":"2022-11-30T04:27:00.424481Z","iopub.execute_input":"2022-11-30T04:27:00.424946Z","iopub.status.idle":"2022-11-30T04:27:00.430192Z","shell.execute_reply.started":"2022-11-30T04:27:00.424909Z","shell.execute_reply":"2022-11-30T04:27:00.429084Z"}}},{"cell_type":"markdown","source":"## Is the data clean?","metadata":{}},{"cell_type":"code","source":"def check_na_values(df):\n    # Append the length of the null rows for each column\n    col_vals = {}\n    for col in df.columns:\n        col_vals[col] = len(df[df[col].isna() == True][col])\n    return col_vals\n        ","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:46:42.768182Z","iopub.execute_input":"2022-11-30T06:46:42.768731Z","iopub.status.idle":"2022-11-30T06:46:42.774959Z","shell.execute_reply.started":"2022-11-30T06:46:42.768698Z","shell.execute_reply":"2022-11-30T06:46:42.773801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_na_values(patient_df)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:46:42.77745Z","iopub.execute_input":"2022-11-30T06:46:42.777998Z","iopub.status.idle":"2022-11-30T06:46:42.845644Z","shell.execute_reply.started":"2022-11-30T06:46:42.777953Z","shell.execute_reply":"2022-11-30T06:46:42.844347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see from this that BIRADS and DENSITY contain the most NA values.\nAge also includes 37 NA values.\nWe can choose to ignore any of these data points or we can selectively use the non NA points. Data requires cleaning or a decision to be made on these columns/datapoints","metadata":{}},{"cell_type":"markdown","source":"## Is the data balanced?","metadata":{}},{"cell_type":"code","source":"cancer_outcomes = [patient_df[\"cancer\"].value_counts()[0], patient_df[\"cancer\"].value_counts()[1]]","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:46:42.847064Z","iopub.execute_input":"2022-11-30T06:46:42.847439Z","iopub.status.idle":"2022-11-30T06:46:42.858061Z","shell.execute_reply.started":"2022-11-30T06:46:42.847407Z","shell.execute_reply":"2022-11-30T06:46:42.856822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.pie(cancer_outcomes, labels=[\"No Cancer\", \"Cancer\"])","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:46:42.859485Z","iopub.execute_input":"2022-11-30T06:46:42.859922Z","iopub.status.idle":"2022-11-30T06:46:42.971907Z","shell.execute_reply.started":"2022-11-30T06:46:42.859891Z","shell.execute_reply":"2022-11-30T06:46:42.970223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This confirms that the data for cancer outcomes are highly unbalanced. We will likely consider upsampling the lower frequency positive occurrences.","metadata":{}},{"cell_type":"code","source":"patient_df[\"laterality\"].value_counts()[\"R\"] / patient_df[\"laterality\"].value_counts()[\"L\"]","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:48:03.654681Z","iopub.execute_input":"2022-11-30T06:48:03.655112Z","iopub.status.idle":"2022-11-30T06:48:03.670327Z","shell.execute_reply.started":"2022-11-30T06:48:03.655076Z","shell.execute_reply":"2022-11-30T06:48:03.669115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"laterality = (patient_df[\"laterality\"].value_counts()[\"L\"], patient_df[\"laterality\"].value_counts()[\"R\"])","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:53:40.541708Z","iopub.execute_input":"2022-11-30T06:53:40.542127Z","iopub.status.idle":"2022-11-30T06:53:40.554899Z","shell.execute_reply.started":"2022-11-30T06:53:40.542093Z","shell.execute_reply":"2022-11-30T06:53:40.553583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.pie(laterality, labels=[\"Left\", \"Right\"], startangle=90)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T06:53:40.967153Z","iopub.execute_input":"2022-11-30T06:53:40.968411Z","iopub.status.idle":"2022-11-30T06:53:41.067297Z","shell.execute_reply.started":"2022-11-30T06:53:40.968359Z","shell.execute_reply":"2022-11-30T06:53:41.065692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Laterality seems to be highly symmetrical with the right side occuring only 6 one-thousandths higher than the left.","metadata":{}},{"cell_type":"code","source":"age_counts = patient_df[\"age\"].value_counts().sort_index()\nage_counts","metadata":{"execution":{"iopub.status.busy":"2022-11-30T07:15:40.590055Z","iopub.execute_input":"2022-11-30T07:15:40.590967Z","iopub.status.idle":"2022-11-30T07:15:40.602499Z","shell.execute_reply.started":"2022-11-30T07:15:40.590918Z","shell.execute_reply":"2022-11-30T07:15:40.601475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar(age_counts.index, age_counts.values)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T07:15:42.888219Z","iopub.execute_input":"2022-11-30T07:15:42.888676Z","iopub.status.idle":"2022-11-30T07:15:43.294426Z","shell.execute_reply.started":"2022-11-30T07:15:42.888638Z","shell.execute_reply":"2022-11-30T07:15:43.293386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age_counts.sort_values()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T07:04:46.296809Z","iopub.execute_input":"2022-11-30T07:04:46.297426Z","iopub.status.idle":"2022-11-30T07:04:46.31508Z","shell.execute_reply.started":"2022-11-30T07:04:46.297363Z","shell.execute_reply":"2022-11-30T07:04:46.313638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This data appears fairly normal, however a Shapiro-Wilks test will be used to confirm if the data is normal.\nOne interesting thing to note is the spike seen in the graph which represents the maximum of the data. This is an interesting finding because it occurs at age 50, while the trend appears to be continuing upward. The only explanation for this could be that 50 is the year when most women begin to receive yearly mammograms. This somewhat arbitrary custom of starting at age 50 is contributing to the skew and the higher incidence of screenings at age 50. \nhttps://www.uspreventiveservicestaskforce.org/uspstf/recommendation/breast-cancer-screening. Just for curiosity I am going to see if this phenomenon occurs with cancer as well at age 50.","metadata":{}},{"cell_type":"code","source":"age_counts_cancer = patient_df[patient_df[\"cancer\"] == True][\"age\"].value_counts().sort_index()\nplt.bar(age_counts_cancer.index, age_counts.values)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T07:15:36.70496Z","iopub.execute_input":"2022-11-30T07:15:36.705418Z","iopub.status.idle":"2022-11-30T07:15:37.14159Z","shell.execute_reply.started":"2022-11-30T07:15:36.705378Z","shell.execute_reply":"2022-11-30T07:15:37.140107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The same phenomenon does not hold true for cancer at age 50. However, you can see that 50 is the local maxima for cancer and is not exceeded until age 56.","metadata":{}},{"cell_type":"markdown","source":"A similar phenonemom could be occuring where younger women are more likely to get screened if they believe they have it, or at a hereditary risk for developing it. This may cause a higher frequency of cancer in younger women in the data. Let's check to see if that occurs.","metadata":{}},{"cell_type":"code","source":"# Create a dataframe for all age counts, and age counts for patients with cancer\nage_counts_df = pd.DataFrame(age_counts)\nage_counts_df.columns = [\"age_counts\"]\nage_counts_cancer_df = pd.DataFrame(age_counts_cancer)\nage_counts_cancer_df.columns = [\"cancer_age_counts\"]\n","metadata":{"execution":{"iopub.status.busy":"2022-11-30T07:25:06.313596Z","iopub.execute_input":"2022-11-30T07:25:06.314031Z","iopub.status.idle":"2022-11-30T07:25:06.321972Z","shell.execute_reply.started":"2022-11-30T07:25:06.313992Z","shell.execute_reply":"2022-11-30T07:25:06.320354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Join the two dataframes on their indices and fill null values with 0\ntotal_age_counts_df = age_counts_df.join(age_counts_cancer_df).fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T07:27:30.791771Z","iopub.execute_input":"2022-11-30T07:27:30.792227Z","iopub.status.idle":"2022-11-30T07:27:30.799668Z","shell.execute_reply.started":"2022-11-30T07:27:30.792187Z","shell.execute_reply":"2022-11-30T07:27:30.798217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a ratio \ntotal_age_counts_df[\"ratio\"] = total_age_counts_df[\"cancer_age_counts\"] / total_age_counts_df[\"age_counts\"]","metadata":{"execution":{"iopub.status.busy":"2022-11-30T07:28:53.609647Z","iopub.execute_input":"2022-11-30T07:28:53.610073Z","iopub.status.idle":"2022-11-30T07:28:53.616515Z","shell.execute_reply.started":"2022-11-30T07:28:53.610035Z","shell.execute_reply":"2022-11-30T07:28:53.615298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar(total_age_counts_df.index, total_age_counts_df[\"ratio\"])","metadata":{"execution":{"iopub.status.busy":"2022-11-30T07:29:25.424747Z","iopub.execute_input":"2022-11-30T07:29:25.425132Z","iopub.status.idle":"2022-11-30T07:29:25.875093Z","shell.execute_reply.started":"2022-11-30T07:29:25.425099Z","shell.execute_reply":"2022-11-30T07:29:25.873386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fortunately there is no phenomenon where younger people in the data are more likely to have cancer","metadata":{}}]}