{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Task**: Predict the presence or absence of cancer in mammography images.","metadata":{}},{"cell_type":"code","source":"!pip install -U pylibjpeg pylibjpeg-openjpeg pylibjpeg-libjpeg pydicom python-gdcm","metadata":{"execution":{"iopub.status.busy":"2023-02-01T06:32:06.928808Z","iopub.execute_input":"2023-02-01T06:32:06.929264Z","iopub.status.idle":"2023-02-01T06:32:18.860965Z","shell.execute_reply.started":"2023-02-01T06:32:06.92922Z","shell.execute_reply":"2023-02-01T06:32:18.859416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# How the data is organized\nThe images are organized in directories by patient id.","metadata":{}},{"cell_type":"code","source":"ls ../input/rsna-breast-cancer-detection/train_images | head -1 -n 4","metadata":{"execution":{"iopub.status.busy":"2023-01-31T08:34:19.110392Z","iopub.execute_input":"2023-01-31T08:34:19.110856Z","iopub.status.idle":"2023-01-31T08:34:21.91527Z","shell.execute_reply.started":"2023-01-31T08:34:19.110805Z","shell.execute_reply":"2023-01-31T08:34:21.914055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Each directory contains several images","metadata":{}},{"cell_type":"code","source":"ls ../input/rsna-breast-cancer-detection/train_images/57175","metadata":{"execution":{"iopub.status.busy":"2023-01-31T08:35:27.594585Z","iopub.execute_input":"2023-01-31T08:35:27.595462Z","iopub.status.idle":"2023-01-31T08:35:28.722295Z","shell.execute_reply.started":"2023-01-31T08:35:27.595411Z","shell.execute_reply":"2023-01-31T08:35:28.720946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 11913 patients in the train set.","metadata":{}},{"cell_type":"code","source":"!ls -l ../input/rsna-breast-cancer-detection/train_images | wc -l \n# wc -l outputs one a count of one row too many when used like this! hence we need to subtract one to get the true count","metadata":{"execution":{"iopub.status.busy":"2023-01-31T08:37:00.519848Z","iopub.execute_input":"2023-01-31T08:37:00.52046Z","iopub.status.idle":"2023-01-31T08:37:04.089929Z","shell.execute_reply.started":"2023-01-31T08:37:00.520401Z","shell.execute_reply":"2023-01-31T08:37:04.088059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In test, we have access to just a single example!","metadata":{}},{"cell_type":"code","source":"ls ../input/rsna-breast-cancer-detection/test_images/10008","metadata":{"execution":{"iopub.status.busy":"2023-01-31T08:37:58.844138Z","iopub.execute_input":"2023-01-31T08:37:58.844586Z","iopub.status.idle":"2023-01-31T08:37:59.963438Z","shell.execute_reply.started":"2023-01-31T08:37:58.844548Z","shell.execute_reply":"2023-01-31T08:37:59.962251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DICOM image format overview\nA DICOM file consists of a header and image data sets packed into a single file. The information within the header is organized as a constant and standardized series of tags. By extracting data from these tags one can access important information regarding the patient demographics, study parameters, etc.","metadata":{}},{"cell_type":"markdown","source":"Let us look at some images.","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\ndef rescale_img_to_hu(dcm_ds):\n    \"\"\"Rescales the image to Hounsfield unit.\"\"\"\n    data = dcm_ds.pixel_array\n    if dcm_ds.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    return data * dcm_ds.RescaleSlope + dcm_ds.RescaleIntercept","metadata":{"execution":{"iopub.status.busy":"2023-01-31T08:53:44.880024Z","iopub.execute_input":"2023-01-31T08:53:44.880441Z","iopub.status.idle":"2023-01-31T08:53:44.887456Z","shell.execute_reply.started":"2023-01-31T08:53:44.880406Z","shell.execute_reply":"2023-01-31T08:53:44.885925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\n\ndef show_images_for_patient(patient_id):\n    patient_dir = os.path.join('../input/rsna-breast-cancer-detection/train_images', str(patient_id))\n    num_images = len(glob.glob(f\"{patient_dir}/*\"))\n    print(f\"Number of images for patient: {num_images}\")\n    fig, axs = plt.subplots(2, 2, figsize=(24,15))\n    axs = axs.flatten()\n    for i, img_path in enumerate(list(Path(patient_dir).iterdir())):\n        ds = pydicom.dcmread(img_path)\n        axs[i].imshow(rescale_img_to_hu(ds), cmap=\"bone\")","metadata":{"execution":{"iopub.status.busy":"2023-01-31T08:53:28.718628Z","iopub.execute_input":"2023-01-31T08:53:28.719448Z","iopub.status.idle":"2023-01-31T08:53:28.726844Z","shell.execute_reply.started":"2023-01-31T08:53:28.719407Z","shell.execute_reply":"2023-01-31T08:53:28.725925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_images_for_patient(10006)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T08:53:47.501069Z","iopub.execute_input":"2023-01-31T08:53:47.501491Z","iopub.status.idle":"2023-01-31T08:54:01.702341Z","shell.execute_reply.started":"2023-01-31T08:53:47.501456Z","shell.execute_reply":"2023-01-31T08:54:01.701422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we see from the plots, the images are of considerable size. We see that the breast occupies only a small portion of the image. We can crop out the part of the image that is not giving any information.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndatasetPath = '/kaggle/input/rsna-breast-cancer-detection'\n\ntrainDfPath = datasetPath + '/train.csv'\ntestDfPath = datasetPath + '/test.csv'\ntrainDf = pd.read_csv(trainDfPath)\ntestDf = pd.read_csv(testDfPath)","metadata":{"execution":{"iopub.status.busy":"2023-02-01T06:32:27.240266Z","iopub.execute_input":"2023-02-01T06:32:27.240755Z","iopub.status.idle":"2023-02-01T06:32:27.334706Z","shell.execute_reply.started":"2023-02-01T06:32:27.240717Z","shell.execute_reply":"2023-02-01T06:32:27.333382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our model should output the likelihood of cancer in the corresponding image. So what are the labels we will train on?","metadata":{}},{"cell_type":"code","source":"print(\"Total train dataset: \" + str(len(trainDf)))\ntrainDf.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T08:24:55.556031Z","iopub.execute_input":"2023-01-31T08:24:55.556432Z","iopub.status.idle":"2023-01-31T08:24:55.586054Z","shell.execute_reply.started":"2023-01-31T08:24:55.556399Z","shell.execute_reply":"2023-01-31T08:24:55.584777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The targets are stored in the train.csv file along with some additional metadata.\n\nThey are stored in the cancer column. Additionally, the file provides mapping of images to laterality (whether an image is of the left or right breast -- this can be very useful in training).","metadata":{}},{"cell_type":"code","source":"print(\"Total test dataset: \" + str(len(testDf)))\ntestDf.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T08:24:58.8206Z","iopub.execute_input":"2023-01-31T08:24:58.821038Z","iopub.status.idle":"2023-01-31T08:24:58.836308Z","shell.execute_reply.started":"2023-01-31T08:24:58.820999Z","shell.execute_reply":"2023-01-31T08:24:58.834894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As expected, the test file doesn't contain the cancer column.\n\nSo how many images do we have to train on?","metadata":{}},{"cell_type":"code","source":"trainDf.shape[0], trainDf.patient_id.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T08:57:24.622349Z","iopub.execute_input":"2023-01-31T08:57:24.623291Z","iopub.status.idle":"2023-01-31T08:57:24.63327Z","shell.execute_reply.started":"2023-01-31T08:57:24.623251Z","shell.execute_reply":"2023-01-31T08:57:24.631915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are a total of 54706 images in train across 11913 patients.","metadata":{}},{"cell_type":"markdown","source":"# Distribution of labels","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\n\nplt.figure(figsize=(20,6))\nplt.subplot(1,2,1)\nax1 = sns.countplot(data=trainDf, x='cancer')\nfor container in ax1.containers:\n    ax1.bar_label(container)\nplt.title('Distribution of targets')","metadata":{"execution":{"iopub.status.busy":"2023-01-31T08:59:41.268323Z","iopub.execute_input":"2023-01-31T08:59:41.269117Z","iopub.status.idle":"2023-01-31T08:59:41.492933Z","shell.execute_reply.started":"2023-01-31T08:59:41.269069Z","shell.execute_reply":"2023-01-31T08:59:41.491513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The classes are highly unbalanced! Most likely, we will need to address this in training","metadata":{}},{"cell_type":"code","source":"import glob\ntrainImages = glob.glob(datasetPath + '/train_images/*/*')","metadata":{"execution":{"iopub.status.busy":"2023-02-01T06:32:45.880555Z","iopub.execute_input":"2023-02-01T06:32:45.881007Z","iopub.status.idle":"2023-02-01T06:32:56.446811Z","shell.execute_reply.started":"2023-02-01T06:32:45.880963Z","shell.execute_reply":"2023-02-01T06:32:56.445571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirName in ['train', 'test']:\n    os.makedirs(dirName, exist_ok = True)","metadata":{"execution":{"iopub.status.busy":"2023-02-01T06:33:05.978933Z","iopub.execute_input":"2023-02-01T06:33:05.979369Z","iopub.status.idle":"2023-02-01T06:33:05.985621Z","shell.execute_reply.started":"2023-02-01T06:33:05.979334Z","shell.execute_reply":"2023-02-01T06:33:05.984239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom multiprocessing import Pool, cpu_count\nimport matplotlib.pyplot as plt\nimport gdcm\n\ndef processDcm(path: str):\n    #Open the DICOM file\n    ds = pydicom.dcmread(path)\n    #Get the image data\n    image = ds.pixel_array\n    #Convert image data to PIL image\n    pilImage = Image.fromarray(image)\n    \n#     print(image.shape, path)\n#     plt.imshow(pilImage, cmap = 'gray')\n\n    patientId = path.split('/')[-2]\n    imageId = path.split('/')[-1].split('.')[0]\n#     print(patientId, imageId)\n    \n    #Resize the image \n    resizedImage = pilImage.resize((256, 256))\n    #Save the image tto a new file\n    resizedImage.save('train/' + str(patientId) + '_' + str(imageId) + '.png')\n    \nwith Pool(cpu_count()) as p:\n    p.map(processDcm, tqdm(trainImages))","metadata":{"execution":{"iopub.status.busy":"2023-02-01T06:33:10.521192Z","iopub.execute_input":"2023-02-01T06:33:10.521607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot images\nprocessedTrainImage = glob.glob('train/*')\n\n#create a figure with a grid of subplots\nfig, axs = plt.subplots(nrows = 3, ncols = 5, figsize = (20, 12))\n\n#plot the image in the subplots\nfor i, ax in enumerate(axs.flat):\n    #get the image and label for current subplot\n#     image, label = data[i], labels[i]\n#     img = image.numpy().transpose((1, 2, 0))\n    image = Image.open(processedTrainImage[i+20])\n#     img = image.numpy().transpose((1, 2, 0))\n#     img = np.clip(img, 0, 1)\n    \n    ax.imshow(image, cmap = 'gray')\n    \n    #set the title of the subplot of the label\n#     ax.set_title(class_names[label])\n\n#show plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-01T06:27:28.450029Z","iopub.execute_input":"2023-02-01T06:27:28.450468Z","iopub.status.idle":"2023-02-01T06:27:30.256702Z","shell.execute_reply.started":"2023-02-01T06:27:28.450434Z","shell.execute_reply":"2023-02-01T06:27:30.255371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}