{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":2021025,"sourceType":"datasetVersion","datasetId":1209633},{"sourceId":2527926,"sourceType":"datasetVersion","datasetId":1531638},{"sourceId":4696088,"sourceType":"datasetVersion","datasetId":2687741},{"sourceId":4874049,"sourceType":"datasetVersion","datasetId":2717932}],"dockerImageVersionId":30302,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Multimodal Medical Imaging Model for Enhanced Diagnostic Predictions\n\n## Introduction\n\n- This notebook presents a basic implementation of a multimodal medical imaging model for enhanced diagnostic predictions.\n- The goal is to leverage multiple imaging modalities to improve the accuracy and reliability of medical diagnoses.\n- By combining information from different modalities, such as MRI, CT scan, and mammography, we can potentially uncover more comprehensive insights and achieve better diagnostic accuracy.\n- The focus is on integrating MRI, CT scan, and mammography modalities to showcase the benefits of a multimodal approach.\n- The implementation includes preprocessing the data, building and training a deep learning model, and evaluating its performance.\n- Visualizations and analysis are provided to interpret the model's predictions and assess its effectiveness.\n- The aim is to provide a starting point for researchers and practitioners interested in multimodal medical imaging data.\n","metadata":{"papermill":{"duration":0.027355,"end_time":"2021-10-24T18:56:44.934968","exception":false,"start_time":"2021-10-24T18:56:44.907613","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import sys\nimport os\nimport platform\nprint(sys.version)\nprint(os.name)\nprint(platform.system())\nprint(platform.release())","metadata":{"papermill":{"duration":0.041954,"end_time":"2021-10-24T18:56:45.005241","exception":false,"start_time":"2021-10-24T18:56:44.963287","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T17:22:33.003463Z","iopub.execute_input":"2024-06-17T17:22:33.003945Z","iopub.status.idle":"2024-06-17T17:22:33.036033Z","shell.execute_reply.started":"2024-06-17T17:22:33.003851Z","shell.execute_reply":"2024-06-17T17:22:33.035078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nis_cuda_enabled = torch.cuda.is_available()\nprint('Cuda enabled', is_cuda_enabled)\nif torch.cuda.is_available():\n    print(torch.cuda.current_device())\n    print(torch.cuda.device(0))\n    print(torch.cuda.device_count())\n    print(torch.cuda.get_device_name(0))","metadata":{"papermill":{"duration":1.43331,"end_time":"2021-10-24T18:56:46.465785","exception":false,"start_time":"2021-10-24T18:56:45.032475","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T18:00:59.91639Z","iopub.execute_input":"2024-06-17T18:00:59.9169Z","iopub.status.idle":"2024-06-17T18:00:59.925871Z","shell.execute_reply.started":"2024-06-17T18:00:59.916854Z","shell.execute_reply":"2024-06-17T18:00:59.924694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvcc --version","metadata":{"papermill":{"duration":0.704667,"end_time":"2021-10-24T18:56:47.197829","exception":false,"start_time":"2021-10-24T18:56:46.493162","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T17:22:46.411407Z","iopub.execute_input":"2024-06-17T17:22:46.412532Z","iopub.status.idle":"2024-06-17T17:22:47.368385Z","shell.execute_reply.started":"2024-06-17T17:22:46.412482Z","shell.execute_reply":"2024-06-17T17:22:47.367158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvidia-smi","metadata":{"papermill":{"duration":0.853962,"end_time":"2021-10-24T18:56:48.07905","exception":false,"start_time":"2021-10-24T18:56:47.225088","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T17:23:51.93174Z","iopub.execute_input":"2024-06-17T17:23:51.932149Z","iopub.status.idle":"2024-06-17T17:23:52.919862Z","shell.execute_reply.started":"2024-06-17T17:23:51.932116Z","shell.execute_reply":"2024-06-17T17:23:52.918664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%reload_ext autoreload\n%autoreload 2\n%matplotlib inline\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport pydicom\nimport pandas as pd\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom tqdm import tqdm\nimport binascii\nfrom PIL import Image\n\nfrom fastai.vision.all import *\nimport numpy as np\nimport pandas as pd\nimport random\nimport os\nnp.set_printoptions(threshold=sys.maxsize)","metadata":{"papermill":{"duration":1.565577,"end_time":"2021-10-24T18:56:49.677142","exception":false,"start_time":"2021-10-24T18:56:48.111565","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T18:01:16.910809Z","iopub.execute_input":"2024-06-17T18:01:16.911195Z","iopub.status.idle":"2024-06-17T18:01:18.285195Z","shell.execute_reply.started":"2024-06-17T18:01:16.91116Z","shell.execute_reply":"2024-06-17T18:01:18.284129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pytorch_lightning as pl\nrandom_seed=1\npl.seed_everything(random_seed) #ultra sound data frame\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T18:01:23.255162Z","iopub.execute_input":"2024-06-17T18:01:23.256361Z","iopub.status.idle":"2024-06-17T18:01:26.159321Z","shell.execute_reply.started":"2024-06-17T18:01:23.256318Z","shell.execute_reply":"2024-06-17T18:01:26.158298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.cuda.empty_cache()","metadata":{"papermill":{"duration":0.067379,"end_time":"2021-10-24T18:56:49.771774","exception":false,"start_time":"2021-10-24T18:56:49.704395","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T18:01:32.052074Z","iopub.execute_input":"2024-06-17T18:01:32.052812Z","iopub.status.idle":"2024-06-17T18:01:32.117953Z","shell.execute_reply.started":"2024-06-17T18:01:32.052774Z","shell.execute_reply":"2024-06-17T18:01:32.116812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Path to the dataset\ndataset_path = '/kaggle/input/breast-ultrasound-images-dataset/Dataset_BUSI_with_GT'\ndataset_append_path = '/input/breast-ultrasound-images-dataset/Dataset_BUSI_with_GT'\nos.makedirs('./ultrasound', exist_ok = True)\nprint('ultrasound foldeepor created')\n\n# Output CSV file\noutput_csv = './ultrasound/ultrasound_data.csv'\n\n# Define possible values for laterality and view\nlaterality_options = ['L', 'R']\nview_options = ['CC', 'MLO']\nages = range(30, 90)  # Assuming age range is between 30 and 90\n\n# Function to generate random patient and site IDs\ndef generate_id():\n    return random.randint(10000, 99999)\n\n# List to store the data\ndata = []\n\n# Iterate over each category folder\nfor category in ['benign', 'malignant', 'normal']:\n    category_path = os.path.join(dataset_path, category)\n    if os.path.exists(category_path):\n        \n        for image_file in os.listdir(category_path):\n            if image_file.endswith('.png'):\n                image_id = os.path.splitext(image_file)[0]\n                site_id = random.choice([1,2,3,4,5])\n                patient_id = generate_id()\n                laterality = random.choice(laterality_options)\n                view = random.choice(view_options)\n                age = random.choice(ages)\n                \n                if category == 'benign':\n                    cancer = 0\n                elif category == 'malignant':\n                    cancer = 1\n                else:\n                    cancer = 0\n#\tsite_id\tpatient_id\timage_id\tlaterality\tview\tage\tcancer\tbiopsy\tinvasive\tBIRADS\timplant\tdensity\tmachine_id\tdifficult_negative_case\n                # Append the row to the data list\n                file = image_file\n                biopsy =0\n                invasive =0\n                BIRADS = float('nan')\n                implant = 0\n                density=float('nan')\n                machine_id = random.choice([1,2,3,4,5])\n                dnf = False\n                \n                data.append([site_id, patient_id, image_id, laterality, view, age, cancer, biopsy,invasive,BIRADS,implant, density, machine_id,dnf, file,dataset_append_path+'/'+category+'/'+file])\n\n# Create a DataFrame\nultrasound_df = pd.DataFrame(data, columns=['site_id', 'patient_id', 'image_id', 'laterality', 'view', 'age', 'cancer','biopsy','invasive','BIRADS','implant','density','machine_id','difficult_negative_case', 'file','path'])\n\n# Save the DataFrame to a CSV file\nultrasound_df.to_csv(output_csv, index=False)\n\nprint(f\"Ultrasound CSV file '{output_csv}' has been created successfully.\")\nultrasound_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T18:01:35.785884Z","iopub.execute_input":"2024-06-17T18:01:35.786569Z","iopub.status.idle":"2024-06-17T18:01:36.345167Z","shell.execute_reply.started":"2024-06-17T18:01:35.78652Z","shell.execute_reply":"2024-06-17T18:01:36.344026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# values distribution\nplt.figure(figsize=(5, 4))\nsns.countplot(data=ultrasound_df, x=\"cancer\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-17T19:40:13.457547Z","iopub.execute_input":"2024-06-17T19:40:13.458969Z","iopub.status.idle":"2024-06-17T19:40:13.776447Z","shell.execute_reply.started":"2024-06-17T19:40:13.458912Z","shell.execute_reply":"2024-06-17T19:40:13.77521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load labels","metadata":{"papermill":{"duration":0.026922,"end_time":"2021-10-24T18:56:49.826788","exception":false,"start_time":"2021-10-24T18:56:49.799866","status":"completed"},"tags":[]}},{"cell_type":"code","source":"EPOCHS = 1\nINPUT_PATH = '/input/rsna-mammography-images-as-pngs/images_as_pngs_512/train_images_processed_512'\nLABELS_PATH = '/kaggle/input/rsna-breast-cancer-detection/train.csv'\nMODEL_EXPORT = 'trained_model'\ndf = pd.read_csv(LABELS_PATH)\n","metadata":{"papermill":{"duration":0.090731,"end_time":"2021-10-24T18:56:49.944474","exception":false,"start_time":"2021-10-24T18:56:49.853743","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T19:07:47.305367Z","iopub.execute_input":"2024-06-17T19:07:47.306269Z","iopub.status.idle":"2024-06-17T19:07:47.437583Z","shell.execute_reply.started":"2024-06-17T19:07:47.306204Z","shell.execute_reply":"2024-06-17T19:07:47.436537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"papermill":{"duration":0.074969,"end_time":"2021-10-24T18:56:50.046629","exception":false,"start_time":"2021-10-24T18:56:49.97166","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-01T07:28:05.343632Z","iopub.execute_input":"2024-06-01T07:28:05.344576Z","iopub.status.idle":"2024-06-01T07:28:05.43024Z","shell.execute_reply.started":"2024-06-01T07:28:05.344525Z","shell.execute_reply":"2024-06-01T07:28:05.429054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# values distribution\nplt.figure(figsize=(5, 4))\nsns.countplot(data=df, x=\"cancer\")\ndata =ultrasound_df\nages = data.groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\ncancer_ages = data[data['cancer'] == 1].groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\nno_cancer_ages = data[data['cancer'] == 0].groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\n\nplt.figure(figsize=(16, 10))\n\nplt.subplot(1, 2, 1)\nsns.histplot(ages, bins=63, color='orange', kde=True)\nplt.title(\"All the patient\")\nplt.xlim(33, 89)\n\nplt.subplot(2, 2, 2)\nsns.histplot(cancer_ages, bins=51, color='red', kde=True)\nplt.title(\"Patients with cancer\")\nplt.xlim(33, 89)\n\nplt.subplot(2, 2, 4)\nsns.histplot(no_cancer_ages, bins=63, color='green', kde=True)\nplt.title(\"Patients without cancer\")\nplt.xlim(33, 89)\n\nplt.suptitle(\"Age distribution of the patients\")\nplt.show()","metadata":{"papermill":{"duration":0.252681,"end_time":"2021-10-24T18:56:50.327818","exception":false,"start_time":"2021-10-24T18:56:50.075137","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T19:40:18.075246Z","iopub.execute_input":"2024-06-17T19:40:18.07604Z","iopub.status.idle":"2024-06-17T19:40:19.406777Z","shell.execute_reply.started":"2024-06-17T19:40:18.076002Z","shell.execute_reply":"2024-06-17T19:40:19.405553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Insights\n- Most of the patients are **older than 40 years old**\n- There is a a **peak at the age of 50**\n- Then, there is a **plateau until 70 years old** before the count of patients drops\n- **Patients with cancer** are usually **older** than the others\n- **Age will be an important feature** for a future model\n* The number of sample images with cancer is very low compared to the number of healthy breast samples. It will be more difficult to make the model learn the desired pattern.\n* All patients with a cancer had a biopsy\n* Only a small percentage of images without cancer resulted in a biopsy","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\nfig, ax = plt.subplots(2, 3, figsize=(16, 8))\nsns.countplot(data['laterality'], palette='Blues_r', ax=ax[0, 0])\nsns.countplot(data['implant'], palette='Greens_r', ax=ax[0, 1])\nsns.countplot(data['difficult_negative_case'], palette='Reds_r', ax=ax[0, 2])\nsns.countplot(data['view'], palette='Oranges_r', ax=ax[1, 0])\nsns.countplot(data['density'], palette='Purples_r', order=['A', 'B', 'C', 'D'], ax=ax[1, 1])\nsns.countplot(data['site_id'], palette='Greys_r', ax=ax[1, 2])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-17T19:45:43.695084Z","iopub.execute_input":"2024-06-17T19:45:43.695519Z","iopub.status.idle":"2024-06-17T19:45:44.411296Z","shell.execute_reply.started":"2024-06-17T19:45:43.695482Z","shell.execute_reply":"2024-06-17T19:45:44.410243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Insights\n* The pictures are balanced in term of laterality\n* Only a few images have implents\n* Some images were difficult to diagnose\n* There are usually only two types of views which are CC and MLO\n* There is a minority of images where the breast tissus density is very large (D) or very low (A). Most of the time, the density is in the middle (B) and (C)\n* Images where taken in two different sites in a balanced way","metadata":{}},{"cell_type":"code","source":"#epo#create output dataset folders\nos.makedirs('./train', exist_ok = True)\nprint('Train folder created')\n\nos.makedirs('./test', exist_ok = True)\nprint('Test folder created')\n\n\nfile_path = '/input/breast-ultrasound-images-dataset/Dataset_BUSI_with_GT/benign/benign (2).png'\n\nif os.path.exists(file_path):\n    print(\"File exists.\")\nelse:\n    print(\"File does not exist.\")","metadata":{"papermill":{"duration":0.071934,"end_time":"2021-10-24T18:56:50.432681","exception":false,"start_time":"2021-10-24T18:56:50.360747","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T18:02:27.497072Z","iopub.execute_input":"2024-06-17T18:02:27.497483Z","iopub.status.idle":"2024-06-17T18:02:27.569005Z","shell.execute_reply.started":"2024-06-17T18:02:27.497444Z","shell.execute_reply":"2024-06-17T18:02:27.568007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['file'] = df.apply(lambda x: f'{x[\"patient_id\"]}/{x[\"image_id\"]}.png', axis=1)\ndf['path'] = INPUT_PATH+'/'+df['file']\nmerged_df= pd.concat([ultrasound_df, df], ignore_index=True)\ndf.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-06-17T18:17:11.504576Z","iopub.execute_input":"2024-06-17T18:17:11.504972Z","iopub.status.idle":"2024-06-17T18:17:12.695042Z","shell.execute_reply.started":"2024-06-17T18:17:11.504936Z","shell.execute_reply":"2024-06-17T18:17:12.694055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"help(ImageDataLoaders.from_df)","metadata":{"execution":{"iopub.status.busy":"2024-06-17T18:17:07.003323Z","iopub.execute_input":"2024-06-17T18:17:07.00374Z","iopub.status.idle":"2024-06-17T18:17:07.07629Z","shell.execute_reply.started":"2024-06-17T18:17:07.003708Z","shell.execute_reply":"2024-06-17T18:17:07.07532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a DataLoaders object is a combination of training and validation data\nimage_data = ImageDataLoaders.from_df(merged_df[['path','cancer']], item_tfms=Resize(224), bs=64, label_col=1, fn_col=0, path='/kaggle')","metadata":{"papermill":{"duration":4.811744,"end_time":"2021-10-24T18:59:08.699444","exception":false,"start_time":"2021-10-24T18:59:03.8877","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T18:02:38.510694Z","iopub.execute_input":"2024-06-17T18:02:38.511078Z","iopub.status.idle":"2024-06-17T18:02:47.343531Z","shell.execute_reply.started":"2024-06-17T18:02:38.511043Z","shell.execute_reply":"2024-06-17T18:02:47.342569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# look at the data\nimage_data.show_batch()","metadata":{"papermill":{"duration":1.041027,"end_time":"2021-10-24T18:59:09.772078","exception":false,"start_time":"2021-10-24T18:59:08.731051","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T18:02:47.345644Z","iopub.execute_input":"2024-06-17T18:02:47.34603Z","iopub.status.idle":"2024-06-17T18:02:49.837168Z","shell.execute_reply.started":"2024-06-17T18:02:47.345992Z","shell.execute_reply":"2024-06-17T18:02:49.836205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training stage","metadata":{"papermill":{"duration":0.033494,"end_time":"2021-10-24T18:59:09.838759","exception":false,"start_time":"2021-10-24T18:59:09.805265","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### DICOM files\nThe image data is in DICOM format. **DICOM (Digital Imaging and Communications in Medicine)** is a standard for storing and transmitting medical images and related information.\n\nIt consists of a set of data elements that are organized into a file structure. These data elements contain information about the medical image, such as the patient's name and medical record number, the image modality (e.g., CT, MRI, X-ray), the date and time the image was taken, and the image itself. The image data can be stored in various formats, such as 8-bit or 16-bit grayscale, or 24-bit color.\n\nIn addition to the image data, the DICOM format also includes metadata that describes the characteristics of the image, such as the image resolution, the size of the image in pixels, and the orientation of the image. This metadata is important for accurately displaying and interpreting the image.\n\nThe DICOM format is widely used in the medical community for storing, sharing, and analyzing medical images. It is supported by a wide range of medical devices, such as scanners, modalities, and workstations, and is used in hospitals, clinics, and research facilities around the world.","metadata":{}},{"cell_type":"code","source":"import torch \nimport torch.nn as nn\nimport torch.nn.functional as F\n\nclass Net(nn.Module):\n    def __init__(self, pretrained=False):\n        super().__init__()\n        # 3 input image channel, 6 output channels, 5x5 square convolution\n        # kernel\n        self.conv1 = nn.Conv2d(3, 6, 5)\n        self.pool = nn.MaxPool2d(2, 2)\n        self.conv2 = nn.Conv2d(6, 16, 5)\n        self.fc1 = nn.Linear(16 * 5 * 5, 120)\n        self.fc2 = nn.Linear(120, 84)\n        self.fc3 = nn.Linear(84, 10)\n\n    def forward(self, x):\n        x = self.pool(F.relu(self.conv1(x)))\n        x = self.pool(F.relu(self.conv2(x)))\n        x = torch.flatten(x, 1) # flatten all dimensions except batch\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = F.relu(self.fc3(x))\n        return x","metadata":{"papermill":{"duration":0.096605,"end_time":"2021-10-24T18:59:09.968261","exception":false,"start_time":"2021-10-24T18:59:09.871656","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T18:17:25.206172Z","iopub.execute_input":"2024-06-17T18:17:25.206632Z","iopub.status.idle":"2024-06-17T18:17:25.278006Z","shell.execute_reply.started":"2024-06-17T18:17:25.206593Z","shell.execute_reply":"2024-06-17T18:17:25.27699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Net()\nprint(model)","metadata":{"papermill":{"duration":0.10169,"end_time":"2021-10-24T18:59:10.103208","exception":false,"start_time":"2021-10-24T18:59:10.001518","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T18:17:37.602063Z","iopub.execute_input":"2024-06-17T18:17:37.602473Z","iopub.status.idle":"2024-06-17T18:17:37.66865Z","shell.execute_reply.started":"2024-06-17T18:17:37.602437Z","shell.execute_reply":"2024-06-17T18:17:37.667462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = list(model.parameters())\nprint(len(params))\nprint(params[0].size())  # conv1's .weight","metadata":{"papermill":{"duration":0.093462,"end_time":"2021-10-24T18:59:10.230611","exception":false,"start_time":"2021-10-24T18:59:10.137149","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T18:17:41.211115Z","iopub.execute_input":"2024-06-17T18:17:41.212106Z","iopub.status.idle":"2024-06-17T18:17:41.283915Z","shell.execute_reply.started":"2024-06-17T18:17:41.212055Z","shell.execute_reply":"2024-06-17T18:17:41.282313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# chooses an appropriate loss function\n#https://docs.fast.ai/metrics.html\nlearn = vision_learner(image_data, Net, metrics=[error_rate, accuracy], model_dir=\"/tmp/model/\").to_fp16()\n# auc_score = RocAuc()\n# f1 = F1Score()\n# learn = vision_learner(image_data, Net, metrics=[auc_score, f1], model_dir=\"/tmp/model/\").to_fp16()","metadata":{"papermill":{"duration":0.191923,"end_time":"2021-10-24T18:59:10.457334","exception":false,"start_time":"2021-10-24T18:59:10.265411","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T18:17:47.462409Z","iopub.execute_input":"2024-06-17T18:17:47.463416Z","iopub.status.idle":"2024-06-17T18:17:47.585247Z","shell.execute_reply.started":"2024-06-17T18:17:47.463369Z","shell.execute_reply":"2024-06-17T18:17:47.584312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.lr_find()","metadata":{"papermill":{"duration":37.266086,"end_time":"2021-10-24T18:59:47.799589","exception":false,"start_time":"2021-10-24T18:59:10.533503","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-01T07:35:30.665689Z","iopub.execute_input":"2024-06-01T07:35:30.6661Z","iopub.status.idle":"2024-06-01T07:36:12.564867Z","shell.execute_reply.started":"2024-06-01T07:35:30.666059Z","shell.execute_reply":"2024-06-01T07:36:12.563448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print model's state_dict\nprint(\"Model's state_dict:\")\nfor param_tensor in model.state_dict():\n    print(param_tensor, \"\\t\", model.state_dict()[param_tensor].size())","metadata":{"papermill":{"duration":0.109149,"end_time":"2021-10-24T18:59:47.945083","exception":false,"start_time":"2021-10-24T18:59:47.835934","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-01T07:36:51.647388Z","iopub.execute_input":"2024-06-01T07:36:51.648403Z","iopub.status.idle":"2024-06-01T07:36:51.724919Z","shell.execute_reply.started":"2024-06-01T07:36:51.648358Z","shell.execute_reply":"2024-06-01T07:36:51.723621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nlearn.fit_one_cycle(EPOCHS, lr_max=1e-2)","metadata":{"papermill":{"duration":29.299863,"end_time":"2021-10-24T19:00:17.28136","exception":false,"start_time":"2021-10-24T18:59:47.981497","status":"completed"},"tags":[],"scrolled":true,"execution":{"iopub.status.busy":"2024-06-17T18:32:50.029513Z","iopub.execute_input":"2024-06-17T18:32:50.029963Z","iopub.status.idle":"2024-06-17T18:39:13.047844Z","shell.execute_reply.started":"2024-06-17T18:32:50.029926Z","shell.execute_reply":"2024-06-17T18:39:13.0465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show results of prediction\nlearn.show_results()","metadata":{"papermill":{"duration":1.677093,"end_time":"2021-10-24T19:00:19.01842","exception":false,"start_time":"2021-10-24T19:00:17.341327","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T18:55:02.796097Z","iopub.execute_input":"2024-06-17T18:55:02.796499Z","iopub.status.idle":"2024-06-17T18:55:04.402123Z","shell.execute_reply.started":"2024-06-17T18:55:02.796463Z","shell.execute_reply":"2024-06-17T18:55:04.401172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Insights\n* It is impossible for me to distinguish by eye an image with cancer from a healthy image\n* on the first image in the second row, we can see that there is discrepancy in the model prediction and actual image status ","metadata":{}},{"cell_type":"code","source":"#save model to disk\nlearn.save(MODEL_EXPORT)","metadata":{"papermill":{"duration":0.104405,"end_time":"2021-10-24T19:00:19.168585","exception":false,"start_time":"2021-10-24T19:00:19.06418","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T19:07:55.701744Z","iopub.execute_input":"2024-06-17T19:07:55.702135Z","iopub.status.idle":"2024-06-17T19:07:55.776266Z","shell.execute_reply.started":"2024-06-17T19:07:55.702099Z","shell.execute_reply":"2024-06-17T19:07:55.775317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interp = ClassificationInterpretation.from_learner(learn)\ninterp.plot_top_losses(9, figsize=(15,11))","metadata":{"papermill":{"duration":1.564351,"end_time":"2021-10-24T19:00:20.773945","exception":false,"start_time":"2021-10-24T19:00:19.209594","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T18:55:36.081609Z","iopub.execute_input":"2024-06-17T18:55:36.082001Z","iopub.status.idle":"2024-06-17T18:56:25.969062Z","shell.execute_reply.started":"2024-06-17T18:55:36.081964Z","shell.execute_reply":"2024-06-17T18:56:25.967961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare test images","metadata":{}},{"cell_type":"code","source":"%%capture\n\n!pip install /kaggle/input/rsnamodules/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl \n\ntry:\n    import pylibjpeg\nexcept:\n   !pip install /kaggle/input/rsna-2022-whl/{pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"execution":{"iopub.status.busy":"2024-06-17T19:08:00.997999Z","iopub.execute_input":"2024-06-17T19:08:00.998434Z","iopub.status.idle":"2024-06-17T19:08:55.463429Z","shell.execute_reply.started":"2024-06-17T19:08:00.998395Z","shell.execute_reply":"2024-06-17T19:08:55.461923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dicomsdl as dicoml\nimport cv2\nimport pydicom\n\nfrom joblib import Parallel, delayed\nimport glob","metadata":{"execution":{"iopub.status.busy":"2024-06-17T19:09:51.84031Z","iopub.execute_input":"2024-06-17T19:09:51.841291Z","iopub.status.idle":"2024-06-17T19:09:52.077344Z","shell.execute_reply.started":"2024-06-17T19:09:51.84122Z","shell.execute_reply":"2024-06-17T19:09:52.076312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom pathlib import Path\nimport numpy as np\nfrom PIL import Image\n\nRESIZE_TO = (512, 512)\n!rm -rf test\n!mkdir test\n\ndef dicom_file_to_ary_v1(path):\n    dicom = pydicom.read_file(path)\n    data = dicom.pixel_array\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\ndef dicom_file_to_ary(path):\n    dicom = dicoml.open(path)\n    data = dicom.pixelData()\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\ndirectories = list(Path('/kaggle/input/rsna-breast-cancer-detection/test_images').iterdir())\n\ndef process_directory(directory_path):\n    parent_directory = str(directory_path).split('/')[-1]\n    !mkdir -p test/{parent_directory}\n    for image_path in directory_path.iterdir():\n        processed_ary = dicom_file_to_ary(str(image_path))\n        im = Image.fromarray(processed_ary).resize(RESIZE_TO)\n        im.save(f'test/{parent_directory}/{image_path.stem}.png')\n        \nimport multiprocessing as mp\n\nwith mp.Pool(64) as p:\n    p.map(process_directory, directories)","metadata":{"execution":{"iopub.status.busy":"2024-06-17T19:09:56.432847Z","iopub.execute_input":"2024-06-17T19:09:56.433682Z","iopub.status.idle":"2024-06-17T19:10:05.436918Z","shell.execute_reply.started":"2024-06-17T19:09:56.433629Z","shell.execute_reply":"2024-06-17T19:10:05.435586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\ndf_test['file'] = df_test.apply(lambda x: f'{x[\"patient_id\"]}/{x[\"image_id\"]}.png', axis=1)\ndf_test.head(5)","metadata":{"papermill":{"duration":0.120071,"end_time":"2021-10-24T19:00:20.941485","exception":false,"start_time":"2021-10-24T19:00:20.821414","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-17T19:19:38.772482Z","iopub.execute_input":"2024-06-17T19:19:38.772921Z","iopub.status.idle":"2024-06-17T19:19:38.865695Z","shell.execute_reply.started":"2024-06-17T19:19:38.77288Z","shell.execute_reply":"2024-06-17T19:19:38.864614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict stage","metadata":{"papermill":{"duration":0.045747,"end_time":"2021-10-24T19:00:21.032654","exception":false,"start_time":"2021-10-24T19:00:20.986907","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# load weights\nlearn = vision_learner(image_data, Net, metrics=[error_rate, accuracy], model_dir=\"/tmp/model/\").to_fp16()\nlearn_new = learn.load(MODEL_EXPORT)","metadata":{"papermill":{"duration":0.102848,"end_time":"2021-10-24T19:00:21.18075","exception":false,"start_time":"2021-10-24T19:00:21.077902","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-01T07:57:36.908306Z","iopub.execute_input":"2024-06-01T07:57:36.908695Z","iopub.status.idle":"2024-06-01T07:57:36.994825Z","shell.execute_reply.started":"2024-06-01T07:57:36.908661Z","shell.execute_reply":"2024-06-01T07:57:36.993961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npreds = []\nfor _, row in df_test.iterrows():\n    full_path = f'./test/{row[\"file\"]}'\n    prediction = learn.predict(full_path)\n    probability = prediction[2][1].item()\n    print(probability)\n    preds.append(probability)","metadata":{"papermill":{"duration":2.291883,"end_time":"2021-10-24T19:00:23.518459","exception":false,"start_time":"2021-10-24T19:00:21.226576","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-01T07:57:41.852164Z","iopub.execute_input":"2024-06-01T07:57:41.852581Z","iopub.status.idle":"2024-06-01T07:57:42.095656Z","shell.execute_reply.started":"2024-06-01T07:57:41.852547Z","shell.execute_reply":"2024-06-01T07:57:42.094546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['cancer']=preds","metadata":{"papermill":{"duration":0.165168,"end_time":"2021-10-24T19:00:23.78773","exception":false,"start_time":"2021-10-24T19:00:23.622562","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-01T07:57:45.592158Z","iopub.execute_input":"2024-06-01T07:57:45.59287Z","iopub.status.idle":"2024-06-01T07:57:45.660298Z","shell.execute_reply.started":"2024-06-01T07:57:45.592832Z","shell.execute_reply":"2024-06-01T07:57:45.659216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_output = df_test[['prediction_id', 'cancer']].groupby('prediction_id').mean().reset_index()\ndf_output.to_csv('submission.csv', index=False)\ndf_output.head()","metadata":{"papermill":{"duration":0.169518,"end_time":"2021-10-24T19:00:24.334656","exception":false,"start_time":"2021-10-24T19:00:24.165138","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-06-01T07:57:56.876994Z","iopub.execute_input":"2024-06-01T07:57:56.877392Z","iopub.status.idle":"2024-06-01T07:57:56.961715Z","shell.execute_reply.started":"2024-06-01T07:57:56.877359Z","shell.execute_reply":"2024-06-01T07:57:56.960705Z"},"trusted":true},"execution_count":null,"outputs":[]}]}