{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":4619805,"sourceType":"datasetVersion","datasetId":2688675},{"sourceId":4976318,"sourceType":"datasetVersion","datasetId":2693468}],"dockerImageVersionId":30302,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### ROI Extractor \n\nDuring the visual data analysis I noticed that there is a large variation in the arrangement of the object in the images. In addition, some objects occupy only a small part of the image. By converting (resizing) without ROI extraction we have a very inefficient use of the reduced image. Most of the picture is blank.\n\n**Solution:**\n- annotate data - I annotated about 500 images in a human in the loop technique (3 models were created - I started from 300 images and ended up about 500) \n- train object detector - I used yolov5 (small - balance between accuracy and speed) \n\n**Result on train DS:**\n* 54601 images processed successfully \n* 105 images - detection failed\n\nModel performance (on my validation DS):\n* mAP@50 -> 0.995      \n* mAP50-95 -> 0.914\n\n<div class=\"alert alert-warning\">If you are interested in:\n    <ul>\n        <li>Annotation dataset</li>\n        <li>yolov5 training notebook</li>\n    </ul>\n    <p>&nbsp;</p>\nLet me know in comment. I will provide it as well in separate notebooks.</div>\n\n**Next steps:**\n* generate ROI based dataset for training (768 pix) -> is available here: https://www.kaggle.com/datasets/remekkinas/rsna-breast-cancer-detection-poi-images\n* use ROI extractor in inference part ","metadata":{}},{"cell_type":"code","source":"%%capture \n\n# Clone yolov5 repository\n!git clone https://github.com/ultralytics/yolov5","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:36:42.002423Z","iopub.execute_input":"2024-03-17T17:36:42.003363Z","iopub.status.idle":"2024-03-17T17:36:43.017221Z","shell.execute_reply.started":"2024-03-17T17:36:42.003321Z","shell.execute_reply":"2024-03-17T17:36:43.015811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import torch\nimport glob\nimport random\nimport cv2\nimport matplotlib.pyplot as plt","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-03-17T17:36:48.337145Z","iopub.execute_input":"2024-03-17T17:36:48.337569Z","iopub.status.idle":"2024-03-17T17:36:48.343335Z","shell.execute_reply.started":"2024-03-17T17:36:48.33753Z","shell.execute_reply":"2024-03-17T17:36:48.342257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TensorFlow libraries\nimport tensorflow as tf\nfrom tensorflow.keras.applications.resnet_v2 import ResNet50V2\nfrom tensorflow.keras.optimizers import RMSprop\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\n\n# basic libraries\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nimport os\n\nimport glob\nfrom glob import glob","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:36:58.212169Z","iopub.execute_input":"2024-03-17T17:36:58.212603Z","iopub.status.idle":"2024-03-17T17:36:58.221427Z","shell.execute_reply.started":"2024-03-17T17:36:58.212564Z","shell.execute_reply":"2024-03-17T17:36:58.220306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load trained model\nmodel = torch.hub.load('./yolov5', 'custom', path='/kaggle/input/rsna-breast-cancer-detection-roi-model/rsna-roi-003.pt', source='local')","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:37:15.122423Z","iopub.execute_input":"2024-03-17T17:37:15.122802Z","iopub.status.idle":"2024-03-17T17:37:26.230795Z","shell.execute_reply.started":"2024-03-17T17:37:15.122772Z","shell.execute_reply":"2024-03-17T17:37:26.229829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the path to the image data\nRSNA_512_path = '/kaggle/input/rsna-breast-cancer-512-pngs'","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:37:35.597502Z","iopub.execute_input":"2024-03-17T17:37:35.598354Z","iopub.status.idle":"2024-03-17T17:37:35.603078Z","shell.execute_reply.started":"2024-03-17T17:37:35.598315Z","shell.execute_reply":"2024-03-17T17:37:35.602014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the csv data.\ndf_train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:37:38.062065Z","iopub.execute_input":"2024-03-17T17:37:38.062517Z","iopub.status.idle":"2024-03-17T17:37:38.159432Z","shell.execute_reply.started":"2024-03-17T17:37:38.06248Z","shell.execute_reply":"2024-03-17T17:37:38.158363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Most of the cases are normal or not-malignant cancer. Thus, physicians sometimes overlook cancer.\ndata = pd.DataFrame(np.concatenate([['Total'] * len(df_train) , ['Maglignant Cancer'] *  len(df_train[df_train['cancer'] == 1]), ['Invasive Cancer'] *  len(df_train[(df_train['cancer'] == 1) & (df_train['invasive'] == 1)])]), columns = [\"class\"])\n\nsns.countplot(x = 'class', data = data)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:37:43.146994Z","iopub.execute_input":"2024-03-17T17:37:43.147886Z","iopub.status.idle":"2024-03-17T17:37:43.280037Z","shell.execute_reply.started":"2024-03-17T17:37:43.14785Z","shell.execute_reply":"2024-03-17T17:37:43.278886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The not-malignant cancer cases were limited into biopsy cases.\nDF_train = df_train[df_train['biopsy'] == 1].reset_index(drop = True)\nDF_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:37:52.471596Z","iopub.execute_input":"2024-03-17T17:37:52.472541Z","iopub.status.idle":"2024-03-17T17:37:52.49418Z","shell.execute_reply.started":"2024-03-17T17:37:52.472504Z","shell.execute_reply":"2024-03-17T17:37:52.493193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The number of positive (malignant) and negative (not-malignat) cases should be the same\n# to create a balanced dataset.\nDF_train = DF_train.groupby(['cancer']).apply(lambda x: x.sample(1158, replace = True)\n                                                      ).reset_index(drop = True)\nprint('New Data Size:', DF_train.shape[0])","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:37:57.533583Z","iopub.execute_input":"2024-03-17T17:37:57.533968Z","iopub.status.idle":"2024-03-17T17:37:57.551309Z","shell.execute_reply.started":"2024-03-17T17:37:57.533937Z","shell.execute_reply":"2024-03-17T17:37:57.550252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the path to each image.\nfor i in range(len(DF_train)):\n    DF_train.loc[i, 'path'] = os.path.join(RSNA_512_path + '/' + str(DF_train.loc[i, 'patient_id']) + '_' + str(DF_train.loc[i, 'image_id']) + '.png')\nDF_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:38:07.732604Z","iopub.execute_input":"2024-03-17T17:38:07.733057Z","iopub.status.idle":"2024-03-17T17:38:08.863347Z","shell.execute_reply.started":"2024-03-17T17:38:07.733021Z","shell.execute_reply":"2024-03-17T17:38:08.862209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# a sample image\nimg = cv2.imread(DF_train.loc[2, 'path'])\nplt.imshow(img, cmap = 'gray')","metadata":{"execution":{"iopub.status.busy":"2024-03-17T18:26:21.716269Z","iopub.execute_input":"2024-03-17T18:26:21.717211Z","iopub.status.idle":"2024-03-17T18:26:21.979787Z","shell.execute_reply.started":"2024-03-17T18:26:21.717174Z","shell.execute_reply":"2024-03-17T18:26:21.978764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normal and cancer images must be equally distrubuted.\ntrain_df, val_df = train_test_split(DF_train, \n                                   test_size = 0.20, \n                                   random_state = 2018,\n                                   stratify = DF_train[['cancer']])\n\nprint('train', train_df.shape[0], 'validation', val_df.shape[0])\nprint('train', train_df['cancer'].value_counts())\nprint('validation', val_df['cancer'].value_counts())\ntrain_df.sample(1)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:40:06.087467Z","iopub.execute_input":"2024-03-17T17:40:06.088243Z","iopub.status.idle":"2024-03-17T17:40:06.128377Z","shell.execute_reply.started":"2024-03-17T17:40:06.088207Z","shell.execute_reply":"2024-03-17T17:40:06.127427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up normal images from the training data.\ntrain_df_normal = train_df[train_df['cancer'] == 0].reset_index(drop = True)\nprint(len(train_df_normal))\ntrain_df_normal.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:40:23.057461Z","iopub.execute_input":"2024-03-17T17:40:23.057866Z","iopub.status.idle":"2024-03-17T17:40:23.080014Z","shell.execute_reply.started":"2024-03-17T17:40:23.057833Z","shell.execute_reply":"2024-03-17T17:40:23.079042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up cancer images from the training data.\ntrain_df_cancer = train_df[train_df['cancer'] == 1].reset_index(drop = True)\nprint(len(train_df_cancer))\ntrain_df_cancer.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:40:24.624237Z","iopub.execute_input":"2024-03-17T17:40:24.624632Z","iopub.status.idle":"2024-03-17T17:40:24.645678Z","shell.execute_reply.started":"2024-03-17T17:40:24.624601Z","shell.execute_reply":"2024-03-17T17:40:24.64465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up normal images from the validation data.\nval_df_normal = val_df[val_df['cancer'] == 0].reset_index(drop = True)\nprint(len(val_df_normal))\nval_df_normal.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:40:30.402074Z","iopub.execute_input":"2024-03-17T17:40:30.402469Z","iopub.status.idle":"2024-03-17T17:40:30.423527Z","shell.execute_reply.started":"2024-03-17T17:40:30.402437Z","shell.execute_reply":"2024-03-17T17:40:30.422331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pick up cancer images from the validation data.\nval_df_cancer = val_df[val_df['cancer'] == 1].reset_index(drop = True)\nprint(len(val_df_cancer))\nval_df_cancer.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:40:52.293093Z","iopub.execute_input":"2024-03-17T17:40:52.293522Z","iopub.status.idle":"2024-03-17T17:40:52.314719Z","shell.execute_reply.started":"2024-03-17T17:40:52.293485Z","shell.execute_reply":"2024-03-17T17:40:52.313588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\n# Define the destination directory.\ndestination_dir = '/kaggle/working/train'\ndestination_dir_sub = '/kaggle/working/train/normal'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in train_df_normal['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:42:11.41734Z","iopub.execute_input":"2024-03-17T17:42:11.418468Z","iopub.status.idle":"2024-03-17T17:42:12.5687Z","shell.execute_reply.started":"2024-03-17T17:42:11.418419Z","shell.execute_reply":"2024-03-17T17:42:12.567804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/train'\ndestination_dir_sub = '/kaggle/working/train/cancer'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in train_df_cancer['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:42:20.497199Z","iopub.execute_input":"2024-03-17T17:42:20.497864Z","iopub.status.idle":"2024-03-17T17:42:21.494112Z","shell.execute_reply.started":"2024-03-17T17:42:20.497828Z","shell.execute_reply":"2024-03-17T17:42:21.493224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/val'\ndestination_dir_sub = '/kaggle/working/val/normal'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in val_df_normal['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:42:27.787224Z","iopub.execute_input":"2024-03-17T17:42:27.788413Z","iopub.status.idle":"2024-03-17T17:42:29.058317Z","shell.execute_reply.started":"2024-03-17T17:42:27.788348Z","shell.execute_reply":"2024-03-17T17:42:29.057356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/val'\ndestination_dir_sub = '/kaggle/working/val/cancer'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\nfor path in val_df_cancer['path']:\n    shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:42:35.022602Z","iopub.execute_input":"2024-03-17T17:42:35.023048Z","iopub.status.idle":"2024-03-17T17:42:35.964133Z","shell.execute_reply.started":"2024-03-17T17:42:35.023012Z","shell.execute_reply":"2024-03-17T17:42:35.963047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nnormal_train_images = glob.glob('/kaggle/working/train/normal/*.png')\ncancer_train_images = glob.glob('/kaggle/working/train/cancer/*.png')","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:42:52.147016Z","iopub.execute_input":"2024-03-17T17:42:52.147648Z","iopub.status.idle":"2024-03-17T17:42:52.159532Z","shell.execute_reply.started":"2024-03-17T17:42:52.147611Z","shell.execute_reply":"2024-03-17T17:42:52.15841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See normal images from the training dataset.\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in enumerate(axes.flat):\n    img = cv2.imread(normal_train_images[i])\n    ax.imshow(img)\n    ax.set_title('Normal')\nfig.tight_layout()    \n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:44:14.325027Z","iopub.execute_input":"2024-03-17T17:44:14.325453Z","iopub.status.idle":"2024-03-17T17:44:15.412343Z","shell.execute_reply.started":"2024-03-17T17:44:14.325419Z","shell.execute_reply":"2024-03-17T17:44:15.411222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See cancer images from the training dataset.\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in enumerate(axes.flat):\n    img = cv2.imread(cancer_train_images[i])\n    ax.imshow(img)\n    ax.set_title('Cancer')\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:44:24.70281Z","iopub.execute_input":"2024-03-17T17:44:24.703195Z","iopub.status.idle":"2024-03-17T17:44:25.435673Z","shell.execute_reply.started":"2024-03-17T17:44:24.703163Z","shell.execute_reply":"2024-03-17T17:44:25.434691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\nimages = []\n\nfor img_file in random.sample(cancer_train_images, 25):  # it is fixed to 25 random predictions - if you want to change it remeber to change plot_roi as well\n    \n    # Read file from file\n    frame = cv2.imread(img_file)\n    \n    # Make prediction\n    detections = model(frame)\n    \n    # Convert results to Pandas style\n    results = detections.pandas().xyxy[0].to_dict(orient=\"records\")\n    \n    # Plot result (in 99.99% it predicts only one instance - certainly you can assure that only best prediction is used)\n    for result in results:\n        images.append(cv2.rectangle(frame, (int(result['xmin']), int(result['ymin'])), (int(result['xmax']), int(result['ymax'])), (255,0,0), 4))\n\n# Plot result\nfig, axes = plt.subplots(5, 5, figsize=(20,20))\n    \nfor idx, image in enumerate(images):\n    i = idx % 5 \n    j = idx // 5 \n    axes[i, j].imshow(image)\n\nplt.subplots_adjust(wspace=0, hspace=.2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:43:22.925238Z","iopub.execute_input":"2024-03-17T17:43:22.925887Z","iopub.status.idle":"2024-03-17T17:43:26.707652Z","shell.execute_reply.started":"2024-03-17T17:43:22.925846Z","shell.execute_reply":"2024-03-17T17:43:26.706673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/ctrain'\ndestination_dir_sub = '/kaggle/working/ctrain/ccancer'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\n# for path in train_df_cancer['path']:\n#     shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:45:59.372812Z","iopub.execute_input":"2024-03-17T17:45:59.373878Z","iopub.status.idle":"2024-03-17T17:45:59.379501Z","shell.execute_reply.started":"2024-03-17T17:45:59.373837Z","shell.execute_reply":"2024-03-17T17:45:59.378436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/ctrain'\ndestination_dir_sub = '/kaggle/working/ctrain/cnormal'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\n# for path in train_df_cancer['path']:\n#     shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:46:04.097269Z","iopub.execute_input":"2024-03-17T17:46:04.097697Z","iopub.status.idle":"2024-03-17T17:46:04.104128Z","shell.execute_reply.started":"2024-03-17T17:46:04.097664Z","shell.execute_reply":"2024-03-17T17:46:04.10306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/cval'\ndestination_dir_sub = '/kaggle/working/cval/cnormal'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\n# for path in val_df_normal['path']:\n#     shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:46:11.813572Z","iopub.execute_input":"2024-03-17T17:46:11.814033Z","iopub.status.idle":"2024-03-17T17:46:11.821043Z","shell.execute_reply.started":"2024-03-17T17:46:11.813998Z","shell.execute_reply":"2024-03-17T17:46:11.81987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the destination directory.\ndestination_dir = '/kaggle/working/cval'\ndestination_dir_sub = '/kaggle/working/cval/ccancer'\n\n# Create the destination directory if it doesn't exist.\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)\n\nif not os.path.exists(destination_dir_sub):\n    os.makedirs(destination_dir_sub)   \n    \n# Copy the images to the destination directory.\n# for path in val_df_normal['path']:\n#     shutil.copy2(path, destination_dir_sub)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:46:32.943062Z","iopub.execute_input":"2024-03-17T17:46:32.944267Z","iopub.status.idle":"2024-03-17T17:46:32.950504Z","shell.execute_reply.started":"2024-03-17T17:46:32.944216Z","shell.execute_reply":"2024-03-17T17:46:32.949545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_directories(dir_paths):\n    for dir_path in dir_paths:\n        if not os.path.exists(dir_path):\n            os.makedirs(dir_path)\n\n# Define directories\nworking_dir = '/kaggle/working'\ntrain_dir = os.path.join(working_dir, 'train')\nval_dir = os.path.join(working_dir, 'val')\nctrain_dir = os.path.join(working_dir, 'ctrain')\ncval_dir = os.path.join(working_dir, 'cval')\nccancer_train_dir = os.path.join(ctrain_dir, 'ccancer')\ncnormal_train_dir = os.path.join(ctrain_dir, 'cnormal')\nccancer_val_dir = os.path.join(cval_dir, 'ccancer')\ncnormal_val_dir = os.path.join(cval_dir, 'cnormal')\n\n\n\ndef detect_and_crop_images(image_paths, output_dir_cancer, output_dir_normal):\n    for image_path in image_paths:\n        # Read the image\n        image = cv2.imread(image_path)\n        \n        # Perform object detection\n        detections = model(image)\n        results = detections.pandas().xyxy[0].to_dict(orient=\"records\")\n        \n        # Create output directory based on cancer or normal\n        if 'cancer' in image_path:\n            output_dir = output_dir_cancer\n        else:\n            output_dir = output_dir_normal\n        \n        # Iterate over each detected object\n        for idx, result in enumerate(results):\n            # Extract bounding box coordinates\n            xmin, ymin, xmax, ymax = int(result['xmin']), int(result['ymin']), int(result['xmax']), int(result['ymax'])\n            \n            # Crop the image\n            cropped_image = image[ymin:ymax, xmin:xmax]\n            \n            # Save the cropped image\n            cv2.imwrite(os.path.join(output_dir, f\"{os.path.basename(image_path)}_{idx}.png\"), cropped_image)\n\n\n# Perform object detection and crop images for training set\ndetect_and_crop_images(glob.glob(os.path.join(train_dir, 'cancer', '*.png')), ccancer_train_dir, cnormal_train_dir)\ndetect_and_crop_images(glob.glob(os.path.join(train_dir, 'normal', '*.png')), ccancer_train_dir, cnormal_train_dir)\n\n# Perform object detection and crop images for validation set\ndetect_and_crop_images(glob.glob(os.path.join(val_dir, 'cancer', '*.png')), ccancer_val_dir, cnormal_val_dir)\ndetect_and_crop_images(glob.glob(os.path.join(val_dir, 'normal', '*.png')), ccancer_val_dir, cnormal_val_dir)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:46:46.168461Z","iopub.execute_input":"2024-03-17T17:46:46.169284Z","iopub.status.idle":"2024-03-17T17:47:26.21092Z","shell.execute_reply.started":"2024-03-17T17:46:46.169242Z","shell.execute_reply":"2024-03-17T17:47:26.209899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = glob.glob('/kaggle/working/ctrain/cnormal/*.png')\nprint(len(images))\n# See normal images from the training dataset.\nfig, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (15, 10), subplot_kw = {'xticks':[], 'yticks':[]})\nfor i, ax in enumerate(axes.flat):\n    img = cv2.imread(images[i])\n    ax.imshow(img)\n    ax.set_title('Normal')\nfig.tight_layout()    \n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:54:02.772174Z","iopub.execute_input":"2024-03-17T17:54:02.773233Z","iopub.status.idle":"2024-03-17T17:54:04.088462Z","shell.execute_reply.started":"2024-03-17T17:54:02.773195Z","shell.execute_reply":"2024-03-17T17:54:04.087463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(rescale = 1./255.,\n                                   zoom_range = 0.2)\nval_datagen = ImageDataGenerator(rescale = 1./255.,)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:54:13.658411Z","iopub.execute_input":"2024-03-17T17:54:13.658795Z","iopub.status.idle":"2024-03-17T17:54:13.664351Z","shell.execute_reply.started":"2024-03-17T17:54:13.658765Z","shell.execute_reply":"2024-03-17T17:54:13.66337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#do not run by hisham!!!!!!!\ntrain_path = '/kaggle/working/train'\nval_path = '/kaggle/working/val'\n\ntrain_generator = train_datagen.flow_from_directory(\n    train_path,\n    target_size = (512, 512),\n    batch_size = 32,\n    class_mode = 'binary'\n)\nvalidation_generator = val_datagen.flow_from_directory(\n        val_path,\n        target_size = (512, 512),\n        batch_size = 16,\n        class_mode = 'binary'\n)","metadata":{"execution":{"iopub.status.busy":"2024-02-18T18:50:15.264866Z","iopub.execute_input":"2024-02-18T18:50:15.26526Z","iopub.status.idle":"2024-02-18T18:50:15.480207Z","shell.execute_reply.started":"2024-02-18T18:50:15.265226Z","shell.execute_reply":"2024-02-18T18:50:15.479227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '/kaggle/working/ctrain'\nval_path = '/kaggle/working/cval'\n\ntrain_generator = train_datagen.flow_from_directory(\n    train_path,\n    target_size = (512, 512),\n    batch_size = 32,\n    class_mode = 'binary'\n)\nvalidation_generator = val_datagen.flow_from_directory(\n        val_path,\n        target_size = (512, 512),\n        batch_size = 16,\n        class_mode = 'binary'\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:54:20.366617Z","iopub.execute_input":"2024-03-17T17:54:20.367053Z","iopub.status.idle":"2024-03-17T17:54:20.583707Z","shell.execute_reply.started":"2024-03-17T17:54:20.367019Z","shell.execute_reply":"2024-03-17T17:54:20.582568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import layers\nfrom tensorflow.keras.layers import Dropout","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:56:37.079779Z","iopub.execute_input":"2024-03-17T17:56:37.080542Z","iopub.status.idle":"2024-03-17T17:56:37.0853Z","shell.execute_reply.started":"2024-03-17T17:56:37.080503Z","shell.execute_reply":"2024-03-17T17:56:37.084217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_classes = 2\n\nmodel = Sequential([\n  layers.experimental.preprocessing.Rescaling(1./255, input_shape=(512, 512, 3)),\n  layers.Conv2D(16, 3, padding='same', activation='relu'),  \n  layers.MaxPooling2D(),   \n  layers.Conv2D(32, 3, padding='same', activation='relu'),\n  layers.MaxPooling2D(),\n  layers.Dropout(0.2),\n  layers.Conv2D(64, 3, padding='same', activation='relu'),\n  layers.MaxPooling2D(),\n  layers.Flatten(),\n  layers.Dense(128, activation='relu'),\n  layers.Dense(num_classes)\n])","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:57:19.044879Z","iopub.execute_input":"2024-03-17T17:57:19.045278Z","iopub.status.idle":"2024-03-17T17:57:19.113574Z","shell.execute_reply.started":"2024-03-17T17:57:19.045245Z","shell.execute_reply":"2024-03-17T17:57:19.112547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-03-17T17:57:23.248181Z","iopub.execute_input":"2024-03-17T17:57:23.249292Z","iopub.status.idle":"2024-03-17T17:57:23.255992Z","shell.execute_reply.started":"2024-03-17T17:57:23.24925Z","shell.execute_reply":"2024-03-17T17:57:23.254796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel.compile(optimizer='adam',\n              loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-03-17T18:01:13.712351Z","iopub.execute_input":"2024-03-17T18:01:13.712892Z","iopub.status.idle":"2024-03-17T18:01:13.732232Z","shell.execute_reply.started":"2024-03-17T18:01:13.712852Z","shell.execute_reply":"2024-03-17T18:01:13.73099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callback = tf.keras.callbacks.EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 4)\nhistory = model.fit(train_generator, validation_data = validation_generator, steps_per_epoch = 20, epochs = 15, callbacks = callback)","metadata":{"execution":{"iopub.status.busy":"2024-03-17T18:01:41.122141Z","iopub.execute_input":"2024-03-17T18:01:41.122558Z","iopub.status.idle":"2024-03-17T18:06:50.433691Z","shell.execute_reply.started":"2024-03-17T18:01:41.122522Z","shell.execute_reply":"2024-03-17T18:06:50.432559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam',\n              loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n              metrics=['accuracy'])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = ResNet50V2(weights = 'imagenet', input_shape = (512, 512, 3), include_top = False)\n\nfor layer in base_model.layers:\n    layer.trainable = False\n    \nmodel = Sequential()\nmodel.add(base_model)\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Dense(128, activation = 'relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(1, activation = 'sigmoid'))\n\nmodel.compile(optimizer = \"adam\", loss = 'binary_crossentropy', metrics = [\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2024-02-18T18:50:30.871638Z","iopub.execute_input":"2024-02-18T18:50:30.872024Z","iopub.status.idle":"2024-02-18T18:50:33.450898Z","shell.execute_reply.started":"2024-02-18T18:50:30.871979Z","shell.execute_reply":"2024-02-18T18:50:33.449961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-02-18T18:50:37.900781Z","iopub.execute_input":"2024-02-18T18:50:37.901838Z","iopub.status.idle":"2024-02-18T18:50:37.917207Z","shell.execute_reply.started":"2024-02-18T18:50:37.901798Z","shell.execute_reply":"2024-02-18T18:50:37.91627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callback = tf.keras.callbacks.EarlyStopping(monitor = \"val_loss\", mode = \"min\", patience = 4)\n\nhistory = model.fit(train_generator, validation_data = validation_generator, steps_per_epoch = 20, epochs = 15, callbacks = callback)","metadata":{"execution":{"iopub.status.busy":"2024-02-18T18:50:50.744677Z","iopub.execute_input":"2024-02-18T18:50:50.745054Z","iopub.status.idle":"2024-02-18T18:55:40.687474Z","shell.execute_reply.started":"2024-02-18T18:50:50.745022Z","shell.execute_reply":"2024-02-18T18:55:40.686422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('mammography_pred_model.h5')","metadata":{"execution":{"iopub.status.busy":"2024-02-18T18:57:30.640562Z","iopub.execute_input":"2024-02-18T18:57:30.641242Z","iopub.status.idle":"2024-02-18T18:57:31.046783Z","shell.execute_reply.started":"2024-02-18T18:57:30.641202Z","shell.execute_reply":"2024-02-18T18:57:31.045564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy = history.history['accuracy']\nval_accuracy = history.history['val_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']","metadata":{"execution":{"iopub.status.busy":"2024-02-18T18:57:34.111252Z","iopub.execute_input":"2024-02-18T18:57:34.112159Z","iopub.status.idle":"2024-02-18T18:57:34.11722Z","shell.execute_reply.started":"2024-02-18T18:57:34.11212Z","shell.execute_reply":"2024-02-18T18:57:34.116248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15,10))\n\nplt.subplot(2, 2, 1)\nplt.plot(accuracy, label = \"Training Accuracy\")\nplt.plot(val_accuracy, label = \"Validation Accuracy\")\nplt.ylim(0.4, 1)\nplt.legend(['Train', 'Validation'], loc = 'upper left')\nplt.title(\"Training vs Validation Accuracy\")\nplt.xlabel('epoch')\nplt.ylabel('accuracy')\n\n\nplt.subplot(2, 2, 2)\nplt.plot(loss, label = \"Training Loss\")\nplt.plot(val_loss, label = \"Validation Loss\")\nplt.legend(['Train', 'Validation'], loc = 'upper left')\nplt.title(\"Training vs Validation Loss\")\nplt.xlabel('epoch')\nplt.ylabel('loss')","metadata":{"execution":{"iopub.status.busy":"2024-02-18T18:57:38.03655Z","iopub.execute_input":"2024-02-18T18:57:38.037477Z","iopub.status.idle":"2024-02-18T18:57:38.427827Z","shell.execute_reply.started":"2024-02-18T18:57:38.037434Z","shell.execute_reply":"2024-02-18T18:57:38.426813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\nmodel = load_model('/kaggle/working/mammography_pred_model.h5')","metadata":{"execution":{"iopub.status.busy":"2024-01-30T19:24:11.700161Z","iopub.execute_input":"2024-01-30T19:24:11.70111Z","iopub.status.idle":"2024-01-30T19:24:13.665662Z","shell.execute_reply.started":"2024-01-30T19:24:11.701068Z","shell.execute_reply":"2024-01-30T19:24:13.66475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(validation_generator)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T19:25:26.255422Z","iopub.execute_input":"2024-01-30T19:25:26.256252Z","iopub.status.idle":"2024-01-30T19:25:31.335742Z","shell.execute_reply.started":"2024-01-30T19:25:26.256191Z","shell.execute_reply":"2024-01-30T19:25:31.334787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred","metadata":{"execution":{"iopub.status.busy":"2024-01-30T19:25:34.220108Z","iopub.execute_input":"2024-01-30T19:25:34.220529Z","iopub.status.idle":"2024-01-30T19:25:34.233519Z","shell.execute_reply.started":"2024-01-30T19:25:34.220494Z","shell.execute_reply":"2024-01-30T19:25:34.232304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = []\nfor prob in pred:\n    if prob >= 0.5:\n        y_pred.append(1)\n    else:\n        y_pred.append(0)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T19:26:01.801541Z","iopub.execute_input":"2024-01-30T19:26:01.801905Z","iopub.status.idle":"2024-01-30T19:26:01.8082Z","shell.execute_reply.started":"2024-01-30T19:26:01.801876Z","shell.execute_reply":"2024-01-30T19:26:01.807079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(y_pred).value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-30T19:26:05.067041Z","iopub.execute_input":"2024-01-30T19:26:05.067979Z","iopub.status.idle":"2024-01-30T19:26:05.076516Z","shell.execute_reply.started":"2024-01-30T19:26:05.067937Z","shell.execute_reply":"2024-01-30T19:26:05.075379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = validation_generator.classes","metadata":{"execution":{"iopub.status.busy":"2024-01-30T19:26:09.567461Z","iopub.execute_input":"2024-01-30T19:26:09.568383Z","iopub.status.idle":"2024-01-30T19:26:09.572522Z","shell.execute_reply.started":"2024-01-30T19:26:09.568346Z","shell.execute_reply":"2024-01-30T19:26:09.571542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"print(y_true)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T19:26:11.99434Z","iopub.execute_input":"2024-01-30T19:26:11.994726Z","iopub.status.idle":"2024-01-30T19:26:12.001863Z","shell.execute_reply.started":"2024-01-30T19:26:11.994693Z","shell.execute_reply":"2024-01-30T19:26:12.000804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(y_true, y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-01-30T19:26:22.939522Z","iopub.execute_input":"2024-01-30T19:26:22.940266Z","iopub.status.idle":"2024-01-30T19:26:22.94633Z","shell.execute_reply.started":"2024-01-30T19:26:22.940221Z","shell.execute_reply":"2024-01-30T19:26:22.945242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the class names.\nclass_names = ['Normal', 'Cancer']\n\n# Create the heatmap with class names as tick labels.\nax = sns.heatmap(cm, annot = True, fmt = '.0f', cmap = \"Blues\", annot_kws = {\"size\": 16},\\\n           xticklabels = class_names, yticklabels = class_names)\n\n# Set the axis labels.\nax.set_xlabel(\"Prediction\")\nax.set_ylabel(\"Truth\")","metadata":{"execution":{"iopub.status.busy":"2024-01-30T19:26:26.326436Z","iopub.execute_input":"2024-01-30T19:26:26.327233Z","iopub.status.idle":"2024-01-30T19:26:26.527576Z","shell.execute_reply.started":"2024-01-30T19:26:26.327179Z","shell.execute_reply":"2024-01-30T19:26:26.52638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_true, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-01-30T19:26:42.992306Z","iopub.execute_input":"2024-01-30T19:26:42.993408Z","iopub.status.idle":"2024-01-30T19:26:43.004472Z","shell.execute_reply.started":"2024-01-30T19:26:42.993366Z","shell.execute_reply":"2024-01-30T19:26:43.003413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load trained model\n# model = torch.hub.load('./yolov5', 'custom', path='/kaggle/input/rsna-breast-cancer-detection-roi-model/rsna-roi-003.pt', source='local')","metadata":{"execution":{"iopub.status.busy":"2022-12-01T09:15:13.888607Z","iopub.execute_input":"2022-12-01T09:15:13.889537Z","iopub.status.idle":"2022-12-01T09:15:19.311387Z","shell.execute_reply.started":"2022-12-01T09:15:13.889498Z","shell.execute_reply":"2022-12-01T09:15:19.310204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create list of input files \n# I use Radek Osmulski dataset -> https://www.kaggle.com/datasets/radek1/rsna-mammography-images-as-pngs \n\n# file_list = glob.glob('/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_768/train_images_processed_768/*/*.png')","metadata":{"execution":{"iopub.status.busy":"2022-12-01T09:15:19.313994Z","iopub.execute_input":"2022-12-01T09:15:19.314895Z","iopub.status.idle":"2022-12-01T09:16:09.854016Z","shell.execute_reply.started":"2022-12-01T09:15:19.314847Z","shell.execute_reply":"2022-12-01T09:16:09.852888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %matplotlib inline\n# images = []\n\n# for img_file in random.sample(file_list, 25):  # it is fixed to 25 random predictions - if you want to change it remeber to change plot_roi as well\n    \n#     # Read file from file\n#     frame = cv2.imread(img_file)\n    \n#     # Make prediction\n#     detections = model(frame)\n    \n#     # Convert results to Pandas style\n#     results = detections.pandas().xyxy[0].to_dict(orient=\"records\")\n    \n#     # Plot result (in 99.99% it predicts only one instance - certainly you can assure that only best prediction is used)\n#     for result in results:\n#         images.append(cv2.rectangle(frame, (int(result['xmin']), int(result['ymin'])), (int(result['xmax']), int(result['ymax'])), (255,0,0), 4))\n\n# # Plot result\n# fig, axes = plt.subplots(5, 5, figsize=(20,20))\n    \n# for idx, image in enumerate(images):\n#     i = idx % 5 \n#     j = idx // 5 \n#     axes[i, j].imshow(image)\n\n# plt.subplots_adjust(wspace=0, hspace=.2)\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-01T09:16:09.865937Z","iopub.execute_input":"2022-12-01T09:16:09.866308Z","iopub.status.idle":"2022-12-01T09:16:19.966626Z","shell.execute_reply.started":"2022-12-01T09:16:09.866273Z","shell.execute_reply":"2022-12-01T09:16:19.965765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thank you! This is my first contribution to RSNA Screening Mammography Breast Cancer Detection.\n\nHave a nice day and competition.","metadata":{}}]}